From 5aec92c9ee9e80ea335ec0e71d1f7858280a4db8 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 14 Feb 2026 14:16:43 +0000 Subject: [PATCH 1/7] Fix excessive Prometheus downsampling for single time series queries The default MAX_GRAPH_POINTS was 100 with a hardcoded default of 60 data points, making graphs look completely different from Grafana. For a 6-hour range, Holmes showed 60 points (1 per 6 min) while Grafana shows ~1440 points at 15s resolution. Changes: - Raise MAX_GRAPH_POINTS default from 100 to 500 - Default step now targets max_points (not hardcoded 60) - Allow LLM to request higher resolution (up to 5x default) for simple single-series queries - Token-based truncation remains as safety net for large responses https://claude.ai/code/session_0174s8zNS3peCtHc2tg9iJ8y Signed-off-by: Claude --- holmes/common/env_vars.py | 2 +- .../plugins/toolsets/prometheus/prometheus.py | 30 ++-- .../grafana/test_grafana_tempo_tools.py | 18 ++- .../plugins/toolsets/test_prometheus_unit.py | 129 ++++++++++++++++++ 4 files changed, 164 insertions(+), 15 deletions(-) diff --git a/holmes/common/env_vars.py b/holmes/common/env_vars.py index 4866700bbc..ba7d1dbff0 100644 --- a/holmes/common/env_vars.py +++ b/holmes/common/env_vars.py @@ -92,7 +92,7 @@ def load_bool(env_var, default: Optional[bool]) -> Optional[bool]: LOG_LLM_USAGE_RESPONSE = load_bool("LOG_LLM_USAGE_RESPONSE", False) -MAX_GRAPH_POINTS = float(os.environ.get("MAX_GRAPH_POINTS", 100)) +MAX_GRAPH_POINTS = float(os.environ.get("MAX_GRAPH_POINTS", 500)) # Limit each tool response to N% of the total context window. # Number between 0 and 100 diff --git a/holmes/plugins/toolsets/prometheus/prometheus.py b/holmes/plugins/toolsets/prometheus/prometheus.py index df91a33c03..3424e17421 100644 --- a/holmes/plugins/toolsets/prometheus/prometheus.py +++ b/holmes/plugins/toolsets/prometheus/prometheus.py @@ -463,23 +463,31 @@ def adjust_step_for_max_points( """ Adjusts the step parameter to ensure the number of data points doesn't exceed max_points. + The default max_points is MAX_GRAPH_POINTS (env var, default 500). The LLM can override + this to request higher resolution (up to 5x the default) for simple queries like single + time series, or lower resolution for overview graphs. Token-based truncation provides + an additional safety net for responses that are too large. + Args: start_timestamp: RFC3339 formatted start time end_timestamp: RFC3339 formatted end time step: The requested step duration in seconds (None for auto-calculation) - max_points_override: Optional override for max points (must be <= MAX_GRAPH_POINTS) + max_points_override: Optional override for max points. Can exceed MAX_GRAPH_POINTS + up to 5x the default to allow higher resolution for simple queries. Returns: Adjusted step value in seconds that ensures points <= max_points """ + hard_limit = MAX_GRAPH_POINTS * 5 + # Use override if provided and valid, otherwise use default max_points = MAX_GRAPH_POINTS if max_points_override is not None: - if max_points_override > MAX_GRAPH_POINTS: + if max_points_override > hard_limit: logging.warning( - f"max_points override ({max_points_override}) exceeds system limit ({MAX_GRAPH_POINTS}), using {MAX_GRAPH_POINTS}" + f"max_points override ({max_points_override}) exceeds hard limit ({hard_limit}), using {hard_limit}" ) - max_points = MAX_GRAPH_POINTS + max_points = hard_limit elif max_points_override < 1: logging.warning( f"max_points override ({max_points_override}) is invalid, using default {MAX_GRAPH_POINTS}" @@ -494,12 +502,11 @@ def adjust_step_for_max_points( time_range_seconds = (end_dt - start_dt).total_seconds() - # If no step provided, calculate a reasonable default - # Aim for ~60 data points across the time range (1 per minute for hourly, etc) + # If no step provided, calculate default targeting max_points data points if step is None: - step = max(1, time_range_seconds / 60) + step = max(1, time_range_seconds / max_points) logging.debug( - f"No step provided, defaulting to {step}s for {time_range_seconds}s range" + f"No step provided, defaulting to {step}s for {time_range_seconds}s range (targeting {max_points} points)" ) current_points = time_range_seconds / step @@ -1562,9 +1569,10 @@ def __init__(self, toolset: "PrometheusToolset"): ), "max_points": ToolParameter( description=( - f"Maximum number of data points to return. Default: {int(MAX_GRAPH_POINTS)}. " - f"Can be reduced to get fewer data points (e.g., 50 for simpler graphs). " - f"Cannot exceed system limit of {int(MAX_GRAPH_POINTS)}. " + f"Maximum number of data points per series. Default: {int(MAX_GRAPH_POINTS)}. " + f"Increase for higher resolution (e.g., {int(MAX_GRAPH_POINTS * 2)} for detailed single-series graphs). " + f"Decrease for overview graphs (e.g., 50). " + f"Maximum: {int(MAX_GRAPH_POINTS * 5)}. " f"If your query would return more points than this limit, the step will be automatically adjusted." ), type="number", diff --git a/tests/plugins/toolsets/grafana/test_grafana_tempo_tools.py b/tests/plugins/toolsets/grafana/test_grafana_tempo_tools.py index 35d07b133b..61c7a31dfe 100644 --- a/tests/plugins/toolsets/grafana/test_grafana_tempo_tools.py +++ b/tests/plugins/toolsets/grafana/test_grafana_tempo_tools.py @@ -551,8 +551,12 @@ def test_all_tools_handle_negative_start(self, tempo_toolset): class TestQueryMetricsRangeWithStepAdjustment: """Test QueryMetricsRange with automatic step adjustment.""" - def test_metrics_range_with_no_step_auto_calculates(self, tempo_toolset): + def test_metrics_range_with_no_step_auto_calculates(self, tempo_toolset, monkeypatch): """Test that step is automatically calculated when not provided.""" + import holmes.plugins.toolsets.grafana.toolset_grafana_tempo as tempo_module + + monkeypatch.setattr(tempo_module, "MAX_GRAPH_POINTS", 100) + tool = QueryMetricsRange(tempo_toolset) with patch( @@ -576,8 +580,12 @@ def test_metrics_range_with_no_step_auto_calculates(self, tempo_toolset): # The function should convert this to "36s" assert kwargs["step"] == "36s" - def test_metrics_range_with_small_step_gets_adjusted(self, tempo_toolset): + def test_metrics_range_with_small_step_gets_adjusted(self, tempo_toolset, monkeypatch): """Test that a too-small step gets adjusted to prevent too many points.""" + import holmes.plugins.toolsets.grafana.toolset_grafana_tempo as tempo_module + + monkeypatch.setattr(tempo_module, "MAX_GRAPH_POINTS", 100) + tool = QueryMetricsRange(tempo_toolset) with patch( @@ -650,8 +658,12 @@ def test_metrics_range_with_bare_number_step(self, tempo_toolset): # Step should be "30s" since 30 seconds is fine for 300 second range assert kwargs["step"] == "30s" - def test_metrics_range_step_adjustment_various_ranges(self, tempo_toolset): + def test_metrics_range_step_adjustment_various_ranges(self, tempo_toolset, monkeypatch): """Test step adjustment for various time ranges.""" + import holmes.plugins.toolsets.grafana.toolset_grafana_tempo as tempo_module + + monkeypatch.setattr(tempo_module, "MAX_GRAPH_POINTS", 100) + tool = QueryMetricsRange(tempo_toolset) test_cases = [ diff --git a/tests/plugins/toolsets/test_prometheus_unit.py b/tests/plugins/toolsets/test_prometheus_unit.py index 3607d7c130..e9c215017a 100644 --- a/tests/plugins/toolsets/test_prometheus_unit.py +++ b/tests/plugins/toolsets/test_prometheus_unit.py @@ -60,3 +60,132 @@ def test_adjust_step_for_max_points( result = adjust_step_for_max_points(start_timestamp, end_timestamp, step) assert result == expected_step + + +@pytest.mark.parametrize( + "start_timestamp, end_timestamp, max_graph_points, expected_step", + [ + # Default step targets max_points data points + # 1 hour range, MAX_GRAPH_POINTS=500 -> step = 3600/500 = 7.2 + ( + "2024-01-01T00:00:00Z", + "2024-01-01T01:00:00Z", + 500, + 7.2, + ), + # 6 hour range, MAX_GRAPH_POINTS=500 -> step = 21600/500 = 43.2 + ( + "2024-01-01T00:00:00Z", + "2024-01-01T06:00:00Z", + 500, + 43.2, + ), + # 24 hour range, MAX_GRAPH_POINTS=500 -> step = 86400/500 = 172.8 + ( + "2024-01-01T00:00:00Z", + "2024-01-02T00:00:00Z", + 500, + 172.8, + ), + # 1 hour range, MAX_GRAPH_POINTS=100 (old default) -> step = 3600/100 = 36 + ( + "2024-01-01T00:00:00Z", + "2024-01-01T01:00:00Z", + 100, + 36.0, + ), + ], +) +def test_default_step_targets_max_points( + monkeypatch, start_timestamp, end_timestamp, max_graph_points, expected_step +): + """When no step is provided, default step should target max_points data points.""" + import holmes.plugins.toolsets.prometheus.prometheus as prom_module + + monkeypatch.setattr(prom_module, "MAX_GRAPH_POINTS", max_graph_points) + + result = adjust_step_for_max_points(start_timestamp, end_timestamp, step=None) + assert result == expected_step + + +class TestMaxPointsOverride: + """Tests for LLM max_points override behavior.""" + + def test_override_above_default_is_allowed(self, monkeypatch): + """LLM can request more points than MAX_GRAPH_POINTS for higher resolution.""" + import holmes.plugins.toolsets.prometheus.prometheus as prom_module + + monkeypatch.setattr(prom_module, "MAX_GRAPH_POINTS", 500.0) + + # 1 hour range, requesting 1000 points -> step = 3600/1000 = 3.6 + result = adjust_step_for_max_points( + "2024-01-01T00:00:00Z", + "2024-01-01T01:00:00Z", + step=None, + max_points_override=1000, + ) + assert result == 3.6 + + def test_override_capped_at_hard_limit(self, monkeypatch): + """Override cannot exceed 5x MAX_GRAPH_POINTS (hard limit).""" + import holmes.plugins.toolsets.prometheus.prometheus as prom_module + + monkeypatch.setattr(prom_module, "MAX_GRAPH_POINTS", 500.0) + + # Hard limit is 500 * 5 = 2500 + # Requesting 5000 should be capped at 2500 + # 1 hour range with 2500 points -> step = 3600/2500 = 1.44 + result = adjust_step_for_max_points( + "2024-01-01T00:00:00Z", + "2024-01-01T01:00:00Z", + step=None, + max_points_override=5000, + ) + assert result == 3600 / 2500 + + def test_override_below_default_is_allowed(self, monkeypatch): + """LLM can request fewer points for simpler graphs.""" + import holmes.plugins.toolsets.prometheus.prometheus as prom_module + + monkeypatch.setattr(prom_module, "MAX_GRAPH_POINTS", 500.0) + + # 1 hour range, requesting only 50 points -> step = 3600/50 = 72 + result = adjust_step_for_max_points( + "2024-01-01T00:00:00Z", + "2024-01-01T01:00:00Z", + step=None, + max_points_override=50, + ) + assert result == 72.0 + + def test_override_invalid_value_uses_default(self, monkeypatch): + """Invalid override (< 1) falls back to default.""" + import holmes.plugins.toolsets.prometheus.prometheus as prom_module + + monkeypatch.setattr(prom_module, "MAX_GRAPH_POINTS", 500.0) + + # Invalid override, should use default of 500 + # 1 hour range with 500 points -> step = 3600/500 = 7.2 + result = adjust_step_for_max_points( + "2024-01-01T00:00:00Z", + "2024-01-01T01:00:00Z", + step=None, + max_points_override=0, + ) + assert result == 7.2 + + def test_override_with_explicit_step_adjusts_if_needed(self, monkeypatch): + """When both step and override are provided, step is adjusted if it exceeds override.""" + import holmes.plugins.toolsets.prometheus.prometheus as prom_module + + monkeypatch.setattr(prom_module, "MAX_GRAPH_POINTS", 500.0) + + # 1 hour range, step=1 (would give 3600 points), max_points=1000 + # 3600 > 1000, so adjusted_step = 3600/1000 = 3.6 + result = adjust_step_for_max_points( + "2024-01-01T00:00:00Z", + "2024-01-01T01:00:00Z", + step=1, + max_points_override=1000, + ) + assert result == 3.6 From a7f7bfd56a244c7a1d5eb76140bbf951b952de8f Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 14 Feb 2026 14:59:51 +0000 Subject: [PATCH 2/7] Reduce max_points hard limit from 5x to 2x to avoid token budget overflows For high-cardinality queries (many time series), a 5x override could generate responses that exceed the tool call token budget. Reducing to 2x still allows meaningful resolution increase for low-cardinality queries while keeping token usage reasonable. The tool description now explicitly guides the LLM to only increase above default for queries returning 1-3 series. https://claude.ai/code/session_0174s8zNS3peCtHc2tg9iJ8y Signed-off-by: Claude --- holmes/plugins/toolsets/prometheus/prometheus.py | 14 +++++++------- tests/plugins/toolsets/test_prometheus_unit.py | 12 ++++++------ 2 files changed, 13 insertions(+), 13 deletions(-) diff --git a/holmes/plugins/toolsets/prometheus/prometheus.py b/holmes/plugins/toolsets/prometheus/prometheus.py index 3424e17421..3a5759a0aa 100644 --- a/holmes/plugins/toolsets/prometheus/prometheus.py +++ b/holmes/plugins/toolsets/prometheus/prometheus.py @@ -464,8 +464,8 @@ def adjust_step_for_max_points( Adjusts the step parameter to ensure the number of data points doesn't exceed max_points. The default max_points is MAX_GRAPH_POINTS (env var, default 500). The LLM can override - this to request higher resolution (up to 5x the default) for simple queries like single - time series, or lower resolution for overview graphs. Token-based truncation provides + this to request higher resolution (up to 2x the default) for simple low-cardinality + queries, or lower resolution for overview graphs. Token-based truncation provides an additional safety net for responses that are too large. Args: @@ -473,12 +473,12 @@ def adjust_step_for_max_points( end_timestamp: RFC3339 formatted end time step: The requested step duration in seconds (None for auto-calculation) max_points_override: Optional override for max points. Can exceed MAX_GRAPH_POINTS - up to 5x the default to allow higher resolution for simple queries. + up to 2x the default to allow higher resolution for low-cardinality queries. Returns: Adjusted step value in seconds that ensures points <= max_points """ - hard_limit = MAX_GRAPH_POINTS * 5 + hard_limit = MAX_GRAPH_POINTS * 2 # Use override if provided and valid, otherwise use default max_points = MAX_GRAPH_POINTS @@ -1570,9 +1570,9 @@ def __init__(self, toolset: "PrometheusToolset"): "max_points": ToolParameter( description=( f"Maximum number of data points per series. Default: {int(MAX_GRAPH_POINTS)}. " - f"Increase for higher resolution (e.g., {int(MAX_GRAPH_POINTS * 2)} for detailed single-series graphs). " - f"Decrease for overview graphs (e.g., 50). " - f"Maximum: {int(MAX_GRAPH_POINTS * 5)}. " + f"Only increase above default for queries returning few time series (1-3 series). " + f"Decrease for overview graphs or high-cardinality queries (e.g., 50). " + f"Maximum: {int(MAX_GRAPH_POINTS * 2)}. " f"If your query would return more points than this limit, the step will be automatically adjusted." ), type="number", diff --git a/tests/plugins/toolsets/test_prometheus_unit.py b/tests/plugins/toolsets/test_prometheus_unit.py index e9c215017a..4e467065cc 100644 --- a/tests/plugins/toolsets/test_prometheus_unit.py +++ b/tests/plugins/toolsets/test_prometheus_unit.py @@ -127,21 +127,21 @@ def test_override_above_default_is_allowed(self, monkeypatch): assert result == 3.6 def test_override_capped_at_hard_limit(self, monkeypatch): - """Override cannot exceed 5x MAX_GRAPH_POINTS (hard limit).""" + """Override cannot exceed 2x MAX_GRAPH_POINTS (hard limit).""" import holmes.plugins.toolsets.prometheus.prometheus as prom_module monkeypatch.setattr(prom_module, "MAX_GRAPH_POINTS", 500.0) - # Hard limit is 500 * 5 = 2500 - # Requesting 5000 should be capped at 2500 - # 1 hour range with 2500 points -> step = 3600/2500 = 1.44 + # Hard limit is 500 * 2 = 1000 + # Requesting 2000 should be capped at 1000 + # 1 hour range with 1000 points -> step = 3600/1000 = 3.6 result = adjust_step_for_max_points( "2024-01-01T00:00:00Z", "2024-01-01T01:00:00Z", step=None, - max_points_override=5000, + max_points_override=2000, ) - assert result == 3600 / 2500 + assert result == 3600 / 1000 def test_override_below_default_is_allowed(self, monkeypatch): """LLM can request fewer points for simpler graphs.""" From 5c3d25228f39d6b4a8b3c870b37a7aca70324fb7 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 14 Feb 2026 15:02:18 +0000 Subject: [PATCH 3/7] Add LLM guidance on query resolution and spike investigation The LLM had no guidance on how resolution affects data quality. Added instructions section explaining: - Default 500 points per series, controllable via max_points - Increase max_points (up to 1000) when investigating spikes/anomalies since they can disappear at lower resolution - Only increase for low-cardinality queries (1-3 series) - Narrow the time range for even higher resolution Also improved the step parameter description. https://claude.ai/code/session_0174s8zNS3peCtHc2tg9iJ8y Signed-off-by: Claude --- holmes/plugins/toolsets/prometheus/prometheus.py | 6 +++++- .../toolsets/prometheus/prometheus_instructions.jinja2 | 7 +++++++ 2 files changed, 12 insertions(+), 1 deletion(-) diff --git a/holmes/plugins/toolsets/prometheus/prometheus.py b/holmes/plugins/toolsets/prometheus/prometheus.py index 3a5759a0aa..f8828641c4 100644 --- a/holmes/plugins/toolsets/prometheus/prometheus.py +++ b/holmes/plugins/toolsets/prometheus/prometheus.py @@ -1549,7 +1549,11 @@ def __init__(self, toolset: "PrometheusToolset"): required=False, ), "step": ToolParameter( - description="Query resolution step width in duration format or float number of seconds", + description=( + "Query resolution step width in duration format or float number of seconds. " + "Smaller step = higher resolution but more data points. " + "If not provided, automatically calculated from the time range and max_points." + ), type="number", required=False, ), diff --git a/holmes/plugins/toolsets/prometheus/prometheus_instructions.jinja2 b/holmes/plugins/toolsets/prometheus/prometheus_instructions.jinja2 index f5621f3c26..4d5ec6db88 100644 --- a/holmes/plugins/toolsets/prometheus/prometheus_instructions.jinja2 +++ b/holmes/plugins/toolsets/prometheus/prometheus_instructions.jinja2 @@ -28,6 +28,13 @@ You are using Coralogix Prometheus. * Be extremely, extremely cautious when answering based on get_label_values because the existence of a label value says NOTHING about the metric value itself (is it high, low, or perhaps the label exists in Prometheus but its an older series not present right now) * DO NOT give answers about metrics based on what 'is typically the case' or 'common knowledge' - if you can't see the actual metric value, you MUST NEVER EVER answer about it - just tell the user your limitations due to the size of the data +## Query Resolution & Data Points +* Range queries return up to 500 data points per series by default. You can control this with the `max_points` parameter. +* When investigating spikes, anomalies, or short-lived events, increase `max_points` up to the maximum (1000) — spikes can completely disappear at lower resolution. +* Only increase `max_points` above default for low-cardinality queries (1-3 time series). For queries returning many series, keep the default or lower it. +* If you need even higher resolution than the maximum allows, narrow the time range instead. For example, zoom into a 30-minute window around the spike rather than querying a full 24-hour range. +* For overview or trend analysis (e.g., "show me the last week"), the default resolution is sufficient. + ## Alert Investigation & Query Execution * When investigating a Prometheus alert, ALWAYS call list_prometheus_rules to get the alert definition * Use Prometheus to query metrics from the alert promql From c84c59a40e903de51ffa817e6e66ce412d935cb0 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 15 Feb 2026 06:27:44 +0000 Subject: [PATCH 4/7] Remove unnecessary overview resolution guidance https://claude.ai/code/session_0174s8zNS3peCtHc2tg9iJ8y Signed-off-by: Claude --- .../plugins/toolsets/prometheus/prometheus_instructions.jinja2 | 1 - 1 file changed, 1 deletion(-) diff --git a/holmes/plugins/toolsets/prometheus/prometheus_instructions.jinja2 b/holmes/plugins/toolsets/prometheus/prometheus_instructions.jinja2 index 4d5ec6db88..ae4ba6a04d 100644 --- a/holmes/plugins/toolsets/prometheus/prometheus_instructions.jinja2 +++ b/holmes/plugins/toolsets/prometheus/prometheus_instructions.jinja2 @@ -33,7 +33,6 @@ You are using Coralogix Prometheus. * When investigating spikes, anomalies, or short-lived events, increase `max_points` up to the maximum (1000) — spikes can completely disappear at lower resolution. * Only increase `max_points` above default for low-cardinality queries (1-3 time series). For queries returning many series, keep the default or lower it. * If you need even higher resolution than the maximum allows, narrow the time range instead. For example, zoom into a 30-minute window around the spike rather than querying a full 24-hour range. -* For overview or trend analysis (e.g., "show me the last week"), the default resolution is sufficient. ## Alert Investigation & Query Execution * When investigating a Prometheus alert, ALWAYS call list_prometheus_rules to get the alert definition From 93025a078a81ab5e2ff53d3a1753467f841acad2 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 15 Feb 2026 16:10:02 +0000 Subject: [PATCH 5/7] Use actual max_points values in LLM instructions instead of hardcoded numbers - Template now uses {{ default_max_points }} and {{ hard_max_points }} from the MAX_GRAPH_POINTS env var so instructions stay accurate if the default changes - Updated max_points tool description: removed "overview graphs" wording, clarified decrease is to avoid hitting the data point limit https://claude.ai/code/session_0174s8zNS3peCtHc2tg9iJ8y Signed-off-by: Claude --- holmes/plugins/toolsets/prometheus/prometheus.py | 14 ++++++++++++-- .../prometheus/prometheus_instructions.jinja2 | 4 ++-- 2 files changed, 14 insertions(+), 4 deletions(-) diff --git a/holmes/plugins/toolsets/prometheus/prometheus.py b/holmes/plugins/toolsets/prometheus/prometheus.py index f8828641c4..c2c025c7ca 100644 --- a/holmes/plugins/toolsets/prometheus/prometheus.py +++ b/holmes/plugins/toolsets/prometheus/prometheus.py @@ -31,6 +31,7 @@ ) from holmes.core.tools_utils.token_counting import count_tool_response_tokens from holmes.core.tools_utils.tool_context_window_limiter import get_pct_token_count +from holmes.plugins.prompts import load_and_render_prompt from holmes.plugins.toolsets.consts import STANDARD_END_DATETIME_TOOL_PARAM_DESCRIPTION from holmes.plugins.toolsets.json_filter_mixin import JsonFilterMixin from holmes.plugins.toolsets.logging_utils.logging_api import ( @@ -1575,7 +1576,7 @@ def __init__(self, toolset: "PrometheusToolset"): description=( f"Maximum number of data points per series. Default: {int(MAX_GRAPH_POINTS)}. " f"Only increase above default for queries returning few time series (1-3 series). " - f"Decrease for overview graphs or high-cardinality queries (e.g., 50). " + f"Decrease for high-cardinality queries (e.g., 50) to avoid hitting maximum number of data points. " f"Maximum: {int(MAX_GRAPH_POINTS * 2)}. " f"If your query would return more points than this limit, the step will be automatically adjusted." ), @@ -1806,7 +1807,16 @@ def _reload_llm_instructions(self): template_file_path = os.path.abspath( os.path.join(os.path.dirname(__file__), "prometheus_instructions.jinja2") ) - self._load_llm_instructions(jinja_template=f"file://{template_file_path}") + tool_names = [t.name for t in self.tools] + self.llm_instructions = load_and_render_prompt( + prompt=f"file://{template_file_path}", + context={ + "tool_names": tool_names, + "config": self.config, + "default_max_points": int(MAX_GRAPH_POINTS), + "hard_max_points": int(MAX_GRAPH_POINTS * 2), + }, + ) def determine_prometheus_class( self, config: dict[str, Any] diff --git a/holmes/plugins/toolsets/prometheus/prometheus_instructions.jinja2 b/holmes/plugins/toolsets/prometheus/prometheus_instructions.jinja2 index ae4ba6a04d..94a329ff83 100644 --- a/holmes/plugins/toolsets/prometheus/prometheus_instructions.jinja2 +++ b/holmes/plugins/toolsets/prometheus/prometheus_instructions.jinja2 @@ -29,8 +29,8 @@ You are using Coralogix Prometheus. * DO NOT give answers about metrics based on what 'is typically the case' or 'common knowledge' - if you can't see the actual metric value, you MUST NEVER EVER answer about it - just tell the user your limitations due to the size of the data ## Query Resolution & Data Points -* Range queries return up to 500 data points per series by default. You can control this with the `max_points` parameter. -* When investigating spikes, anomalies, or short-lived events, increase `max_points` up to the maximum (1000) — spikes can completely disappear at lower resolution. +* Range queries return up to {{ default_max_points }} data points per series by default. You can control this with the `max_points` parameter. +* When investigating spikes, anomalies, or short-lived events, increase `max_points` up to the maximum ({{ hard_max_points }}) — spikes can completely disappear at lower resolution. * Only increase `max_points` above default for low-cardinality queries (1-3 time series). For queries returning many series, keep the default or lower it. * If you need even higher resolution than the maximum allows, narrow the time range instead. For example, zoom into a 30-minute window around the spike rather than querying a full 24-hour range. From 4b5c604675826746a6a751483e4c08fbb6a62639 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 15 Feb 2026 16:22:07 +0000 Subject: [PATCH 6/7] Extract hard limit into MAX_GRAPH_POINTS_HARD_LIMIT env var Co-locate with MAX_GRAPH_POINTS for better locality. Defaults to MAX_GRAPH_POINTS * 2 but can now be configured independently. https://claude.ai/code/session_0174s8zNS3peCtHc2tg9iJ8y Signed-off-by: Claude --- holmes/common/env_vars.py | 1 + holmes/plugins/toolsets/prometheus/prometheus.py | 8 ++++---- 2 files changed, 5 insertions(+), 4 deletions(-) diff --git a/holmes/common/env_vars.py b/holmes/common/env_vars.py index ba7d1dbff0..0b91859386 100644 --- a/holmes/common/env_vars.py +++ b/holmes/common/env_vars.py @@ -93,6 +93,7 @@ def load_bool(env_var, default: Optional[bool]) -> Optional[bool]: MAX_GRAPH_POINTS = float(os.environ.get("MAX_GRAPH_POINTS", 500)) +MAX_GRAPH_POINTS_HARD_LIMIT = float(os.environ.get("MAX_GRAPH_POINTS_HARD_LIMIT", MAX_GRAPH_POINTS * 2)) # Limit each tool response to N% of the total context window. # Number between 0 and 100 diff --git a/holmes/plugins/toolsets/prometheus/prometheus.py b/holmes/plugins/toolsets/prometheus/prometheus.py index c2c025c7ca..051bc1ae7c 100644 --- a/holmes/plugins/toolsets/prometheus/prometheus.py +++ b/holmes/plugins/toolsets/prometheus/prometheus.py @@ -17,7 +17,7 @@ from requests import RequestException from requests.exceptions import SSLError # type: ignore -from holmes.common.env_vars import IS_OPENSHIFT, MAX_GRAPH_POINTS +from holmes.common.env_vars import IS_OPENSHIFT, MAX_GRAPH_POINTS, MAX_GRAPH_POINTS_HARD_LIMIT from holmes.common.openshift import load_openshift_token from holmes.core.tools import ( CallablePrerequisite, @@ -479,7 +479,7 @@ def adjust_step_for_max_points( Returns: Adjusted step value in seconds that ensures points <= max_points """ - hard_limit = MAX_GRAPH_POINTS * 2 + hard_limit = MAX_GRAPH_POINTS_HARD_LIMIT # Use override if provided and valid, otherwise use default max_points = MAX_GRAPH_POINTS @@ -1577,7 +1577,7 @@ def __init__(self, toolset: "PrometheusToolset"): f"Maximum number of data points per series. Default: {int(MAX_GRAPH_POINTS)}. " f"Only increase above default for queries returning few time series (1-3 series). " f"Decrease for high-cardinality queries (e.g., 50) to avoid hitting maximum number of data points. " - f"Maximum: {int(MAX_GRAPH_POINTS * 2)}. " + f"Maximum: {int(MAX_GRAPH_POINTS_HARD_LIMIT)}. " f"If your query would return more points than this limit, the step will be automatically adjusted." ), type="number", @@ -1814,7 +1814,7 @@ def _reload_llm_instructions(self): "tool_names": tool_names, "config": self.config, "default_max_points": int(MAX_GRAPH_POINTS), - "hard_max_points": int(MAX_GRAPH_POINTS * 2), + "hard_max_points": int(MAX_GRAPH_POINTS_HARD_LIMIT), }, ) From 22cd0c0ad7366f1c530985ebe112f6b3c88997df Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 16 Feb 2026 11:26:53 +0000 Subject: [PATCH 7/7] Lower default MAX_GRAPH_POINTS from 500 to 300 Also fix tests to monkeypatch MAX_GRAPH_POINTS_HARD_LIMIT alongside MAX_GRAPH_POINTS so they remain independent of the default values. https://claude.ai/code/session_0174s8zNS3peCtHc2tg9iJ8y Signed-off-by: Claude --- holmes/common/env_vars.py | 2 +- tests/plugins/toolsets/test_prometheus_unit.py | 5 ++++- 2 files changed, 5 insertions(+), 2 deletions(-) diff --git a/holmes/common/env_vars.py b/holmes/common/env_vars.py index 0b91859386..51390a820a 100644 --- a/holmes/common/env_vars.py +++ b/holmes/common/env_vars.py @@ -92,7 +92,7 @@ def load_bool(env_var, default: Optional[bool]) -> Optional[bool]: LOG_LLM_USAGE_RESPONSE = load_bool("LOG_LLM_USAGE_RESPONSE", False) -MAX_GRAPH_POINTS = float(os.environ.get("MAX_GRAPH_POINTS", 500)) +MAX_GRAPH_POINTS = float(os.environ.get("MAX_GRAPH_POINTS", 300)) MAX_GRAPH_POINTS_HARD_LIMIT = float(os.environ.get("MAX_GRAPH_POINTS_HARD_LIMIT", MAX_GRAPH_POINTS * 2)) # Limit each tool response to N% of the total context window. diff --git a/tests/plugins/toolsets/test_prometheus_unit.py b/tests/plugins/toolsets/test_prometheus_unit.py index 4e467065cc..6c3ce7218a 100644 --- a/tests/plugins/toolsets/test_prometheus_unit.py +++ b/tests/plugins/toolsets/test_prometheus_unit.py @@ -116,6 +116,7 @@ def test_override_above_default_is_allowed(self, monkeypatch): import holmes.plugins.toolsets.prometheus.prometheus as prom_module monkeypatch.setattr(prom_module, "MAX_GRAPH_POINTS", 500.0) + monkeypatch.setattr(prom_module, "MAX_GRAPH_POINTS_HARD_LIMIT", 1000.0) # 1 hour range, requesting 1000 points -> step = 3600/1000 = 3.6 result = adjust_step_for_max_points( @@ -127,10 +128,11 @@ def test_override_above_default_is_allowed(self, monkeypatch): assert result == 3.6 def test_override_capped_at_hard_limit(self, monkeypatch): - """Override cannot exceed 2x MAX_GRAPH_POINTS (hard limit).""" + """Override cannot exceed MAX_GRAPH_POINTS_HARD_LIMIT.""" import holmes.plugins.toolsets.prometheus.prometheus as prom_module monkeypatch.setattr(prom_module, "MAX_GRAPH_POINTS", 500.0) + monkeypatch.setattr(prom_module, "MAX_GRAPH_POINTS_HARD_LIMIT", 1000.0) # Hard limit is 500 * 2 = 1000 # Requesting 2000 should be capped at 1000 @@ -179,6 +181,7 @@ def test_override_with_explicit_step_adjusts_if_needed(self, monkeypatch): import holmes.plugins.toolsets.prometheus.prometheus as prom_module monkeypatch.setattr(prom_module, "MAX_GRAPH_POINTS", 500.0) + monkeypatch.setattr(prom_module, "MAX_GRAPH_POINTS_HARD_LIMIT", 1000.0) # 1 hour range, step=1 (would give 3600 points), max_points=1000 # 3600 > 1000, so adjusted_step = 3600/1000 = 3.6