Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion holmes/common/env_vars.py
Original file line number Diff line number Diff line change
Expand Up @@ -92,7 +92,8 @@ def load_bool(env_var, default: Optional[bool]) -> Optional[bool]:
LOG_LLM_USAGE_RESPONSE = load_bool("LOG_LLM_USAGE_RESPONSE", False)


MAX_GRAPH_POINTS = float(os.environ.get("MAX_GRAPH_POINTS", 100))
MAX_GRAPH_POINTS = float(os.environ.get("MAX_GRAPH_POINTS", 300))
MAX_GRAPH_POINTS_HARD_LIMIT = float(os.environ.get("MAX_GRAPH_POINTS_HARD_LIMIT", MAX_GRAPH_POINTS * 2))

# Limit each tool response to N% of the total context window.
# Number between 0 and 100
Expand Down
50 changes: 36 additions & 14 deletions holmes/plugins/toolsets/prometheus/prometheus.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,7 +17,7 @@
from requests import RequestException
from requests.exceptions import SSLError # type: ignore

from holmes.common.env_vars import IS_OPENSHIFT, MAX_GRAPH_POINTS
from holmes.common.env_vars import IS_OPENSHIFT, MAX_GRAPH_POINTS, MAX_GRAPH_POINTS_HARD_LIMIT
from holmes.common.openshift import load_openshift_token
from holmes.core.tools import (
CallablePrerequisite,
Expand All @@ -31,6 +31,7 @@
)
from holmes.core.tools_utils.token_counting import count_tool_response_tokens
from holmes.core.tools_utils.tool_context_window_limiter import get_pct_token_count
from holmes.plugins.prompts import load_and_render_prompt
from holmes.plugins.toolsets.consts import STANDARD_END_DATETIME_TOOL_PARAM_DESCRIPTION
from holmes.plugins.toolsets.json_filter_mixin import JsonFilterMixin
from holmes.plugins.toolsets.logging_utils.logging_api import (
Expand Down Expand Up @@ -463,23 +464,31 @@ def adjust_step_for_max_points(
"""
Adjusts the step parameter to ensure the number of data points doesn't exceed max_points.

The default max_points is MAX_GRAPH_POINTS (env var, default 500). The LLM can override
this to request higher resolution (up to 2x the default) for simple low-cardinality
queries, or lower resolution for overview graphs. Token-based truncation provides
an additional safety net for responses that are too large.

Args:
start_timestamp: RFC3339 formatted start time
end_timestamp: RFC3339 formatted end time
step: The requested step duration in seconds (None for auto-calculation)
max_points_override: Optional override for max points (must be <= MAX_GRAPH_POINTS)
max_points_override: Optional override for max points. Can exceed MAX_GRAPH_POINTS
up to 2x the default to allow higher resolution for low-cardinality queries.

Returns:
Adjusted step value in seconds that ensures points <= max_points
"""
hard_limit = MAX_GRAPH_POINTS_HARD_LIMIT

# Use override if provided and valid, otherwise use default
max_points = MAX_GRAPH_POINTS
if max_points_override is not None:
if max_points_override > MAX_GRAPH_POINTS:
if max_points_override > hard_limit:
logging.warning(
f"max_points override ({max_points_override}) exceeds system limit ({MAX_GRAPH_POINTS}), using {MAX_GRAPH_POINTS}"
f"max_points override ({max_points_override}) exceeds hard limit ({hard_limit}), using {hard_limit}"
)
max_points = MAX_GRAPH_POINTS
max_points = hard_limit
elif max_points_override < 1:
logging.warning(
f"max_points override ({max_points_override}) is invalid, using default {MAX_GRAPH_POINTS}"
Expand All @@ -494,12 +503,11 @@ def adjust_step_for_max_points(

time_range_seconds = (end_dt - start_dt).total_seconds()

# If no step provided, calculate a reasonable default
# Aim for ~60 data points across the time range (1 per minute for hourly, etc)
# If no step provided, calculate default targeting max_points data points
if step is None:
step = max(1, time_range_seconds / 60)
step = max(1, time_range_seconds / max_points)
logging.debug(
f"No step provided, defaulting to {step}s for {time_range_seconds}s range"
f"No step provided, defaulting to {step}s for {time_range_seconds}s range (targeting {max_points} points)"
)

current_points = time_range_seconds / step
Expand Down Expand Up @@ -1542,7 +1550,11 @@ def __init__(self, toolset: "PrometheusToolset"):
required=False,
),
"step": ToolParameter(
description="Query resolution step width in duration format or float number of seconds",
description=(
"Query resolution step width in duration format or float number of seconds. "
"Smaller step = higher resolution but more data points. "
"If not provided, automatically calculated from the time range and max_points."
),
type="number",
required=False,
),
Expand All @@ -1562,9 +1574,10 @@ def __init__(self, toolset: "PrometheusToolset"):
),
"max_points": ToolParameter(
description=(
f"Maximum number of data points to return. Default: {int(MAX_GRAPH_POINTS)}. "
f"Can be reduced to get fewer data points (e.g., 50 for simpler graphs). "
f"Cannot exceed system limit of {int(MAX_GRAPH_POINTS)}. "
f"Maximum number of data points per series. Default: {int(MAX_GRAPH_POINTS)}. "
f"Only increase above default for queries returning few time series (1-3 series). "
f"Decrease for high-cardinality queries (e.g., 50) to avoid hitting maximum number of data points. "
f"Maximum: {int(MAX_GRAPH_POINTS_HARD_LIMIT)}. "
f"If your query would return more points than this limit, the step will be automatically adjusted."
),
type="number",
Expand Down Expand Up @@ -1794,7 +1807,16 @@ def _reload_llm_instructions(self):
template_file_path = os.path.abspath(
os.path.join(os.path.dirname(__file__), "prometheus_instructions.jinja2")
)
self._load_llm_instructions(jinja_template=f"file://{template_file_path}")
tool_names = [t.name for t in self.tools]
self.llm_instructions = load_and_render_prompt(
prompt=f"file://{template_file_path}",
context={
"tool_names": tool_names,
"config": self.config,
"default_max_points": int(MAX_GRAPH_POINTS),
"hard_max_points": int(MAX_GRAPH_POINTS_HARD_LIMIT),
},
)

def determine_prometheus_class(
self, config: dict[str, Any]
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -28,6 +28,12 @@ You are using Coralogix Prometheus.
* Be extremely, extremely cautious when answering based on get_label_values because the existence of a label value says NOTHING about the metric value itself (is it high, low, or perhaps the label exists in Prometheus but its an older series not present right now)
* DO NOT give answers about metrics based on what 'is typically the case' or 'common knowledge' - if you can't see the actual metric value, you MUST NEVER EVER answer about it - just tell the user your limitations due to the size of the data

## Query Resolution & Data Points
* Range queries return up to {{ default_max_points }} data points per series by default. You can control this with the `max_points` parameter.
* When investigating spikes, anomalies, or short-lived events, increase `max_points` up to the maximum ({{ hard_max_points }}) — spikes can completely disappear at lower resolution.
* Only increase `max_points` above default for low-cardinality queries (1-3 time series). For queries returning many series, keep the default or lower it.
* If you need even higher resolution than the maximum allows, narrow the time range instead. For example, zoom into a 30-minute window around the spike rather than querying a full 24-hour range.

## Alert Investigation & Query Execution
* When investigating a Prometheus alert, ALWAYS call list_prometheus_rules to get the alert definition
* Use Prometheus to query metrics from the alert promql
Expand Down
18 changes: 15 additions & 3 deletions tests/plugins/toolsets/grafana/test_grafana_tempo_tools.py
Original file line number Diff line number Diff line change
Expand Up @@ -551,8 +551,12 @@ def test_all_tools_handle_negative_start(self, tempo_toolset):
class TestQueryMetricsRangeWithStepAdjustment:
"""Test QueryMetricsRange with automatic step adjustment."""

def test_metrics_range_with_no_step_auto_calculates(self, tempo_toolset):
def test_metrics_range_with_no_step_auto_calculates(self, tempo_toolset, monkeypatch):
"""Test that step is automatically calculated when not provided."""
import holmes.plugins.toolsets.grafana.toolset_grafana_tempo as tempo_module

monkeypatch.setattr(tempo_module, "MAX_GRAPH_POINTS", 100)

tool = QueryMetricsRange(tempo_toolset)

with patch(
Expand All @@ -576,8 +580,12 @@ def test_metrics_range_with_no_step_auto_calculates(self, tempo_toolset):
# The function should convert this to "36s"
assert kwargs["step"] == "36s"

def test_metrics_range_with_small_step_gets_adjusted(self, tempo_toolset):
def test_metrics_range_with_small_step_gets_adjusted(self, tempo_toolset, monkeypatch):
"""Test that a too-small step gets adjusted to prevent too many points."""
import holmes.plugins.toolsets.grafana.toolset_grafana_tempo as tempo_module

monkeypatch.setattr(tempo_module, "MAX_GRAPH_POINTS", 100)

tool = QueryMetricsRange(tempo_toolset)

with patch(
Expand Down Expand Up @@ -650,8 +658,12 @@ def test_metrics_range_with_bare_number_step(self, tempo_toolset):
# Step should be "30s" since 30 seconds is fine for 300 second range
assert kwargs["step"] == "30s"

def test_metrics_range_step_adjustment_various_ranges(self, tempo_toolset):
def test_metrics_range_step_adjustment_various_ranges(self, tempo_toolset, monkeypatch):
"""Test step adjustment for various time ranges."""
import holmes.plugins.toolsets.grafana.toolset_grafana_tempo as tempo_module

monkeypatch.setattr(tempo_module, "MAX_GRAPH_POINTS", 100)

tool = QueryMetricsRange(tempo_toolset)

test_cases = [
Expand Down
132 changes: 132 additions & 0 deletions tests/plugins/toolsets/test_prometheus_unit.py
Original file line number Diff line number Diff line change
Expand Up @@ -60,3 +60,135 @@ def test_adjust_step_for_max_points(

result = adjust_step_for_max_points(start_timestamp, end_timestamp, step)
assert result == expected_step


@pytest.mark.parametrize(
"start_timestamp, end_timestamp, max_graph_points, expected_step",
[
# Default step targets max_points data points
# 1 hour range, MAX_GRAPH_POINTS=500 -> step = 3600/500 = 7.2
(
"2024-01-01T00:00:00Z",
"2024-01-01T01:00:00Z",
500,
7.2,
),
# 6 hour range, MAX_GRAPH_POINTS=500 -> step = 21600/500 = 43.2
(
"2024-01-01T00:00:00Z",
"2024-01-01T06:00:00Z",
500,
43.2,
),
# 24 hour range, MAX_GRAPH_POINTS=500 -> step = 86400/500 = 172.8
(
"2024-01-01T00:00:00Z",
"2024-01-02T00:00:00Z",
500,
172.8,
),
# 1 hour range, MAX_GRAPH_POINTS=100 (old default) -> step = 3600/100 = 36
(
"2024-01-01T00:00:00Z",
"2024-01-01T01:00:00Z",
100,
36.0,
),
],
)
def test_default_step_targets_max_points(
monkeypatch, start_timestamp, end_timestamp, max_graph_points, expected_step
):
"""When no step is provided, default step should target max_points data points."""
import holmes.plugins.toolsets.prometheus.prometheus as prom_module

monkeypatch.setattr(prom_module, "MAX_GRAPH_POINTS", max_graph_points)

result = adjust_step_for_max_points(start_timestamp, end_timestamp, step=None)
assert result == expected_step


class TestMaxPointsOverride:
"""Tests for LLM max_points override behavior."""

def test_override_above_default_is_allowed(self, monkeypatch):
"""LLM can request more points than MAX_GRAPH_POINTS for higher resolution."""
import holmes.plugins.toolsets.prometheus.prometheus as prom_module

monkeypatch.setattr(prom_module, "MAX_GRAPH_POINTS", 500.0)
monkeypatch.setattr(prom_module, "MAX_GRAPH_POINTS_HARD_LIMIT", 1000.0)

# 1 hour range, requesting 1000 points -> step = 3600/1000 = 3.6
result = adjust_step_for_max_points(
"2024-01-01T00:00:00Z",
"2024-01-01T01:00:00Z",
step=None,
max_points_override=1000,
)
assert result == 3.6

def test_override_capped_at_hard_limit(self, monkeypatch):
"""Override cannot exceed MAX_GRAPH_POINTS_HARD_LIMIT."""
import holmes.plugins.toolsets.prometheus.prometheus as prom_module

monkeypatch.setattr(prom_module, "MAX_GRAPH_POINTS", 500.0)
monkeypatch.setattr(prom_module, "MAX_GRAPH_POINTS_HARD_LIMIT", 1000.0)

# Hard limit is 500 * 2 = 1000
# Requesting 2000 should be capped at 1000
# 1 hour range with 1000 points -> step = 3600/1000 = 3.6
result = adjust_step_for_max_points(
"2024-01-01T00:00:00Z",
"2024-01-01T01:00:00Z",
step=None,
max_points_override=2000,
)
assert result == 3600 / 1000

def test_override_below_default_is_allowed(self, monkeypatch):
"""LLM can request fewer points for simpler graphs."""
import holmes.plugins.toolsets.prometheus.prometheus as prom_module

monkeypatch.setattr(prom_module, "MAX_GRAPH_POINTS", 500.0)

# 1 hour range, requesting only 50 points -> step = 3600/50 = 72
result = adjust_step_for_max_points(
"2024-01-01T00:00:00Z",
"2024-01-01T01:00:00Z",
step=None,
max_points_override=50,
)
assert result == 72.0

def test_override_invalid_value_uses_default(self, monkeypatch):
"""Invalid override (< 1) falls back to default."""
import holmes.plugins.toolsets.prometheus.prometheus as prom_module

monkeypatch.setattr(prom_module, "MAX_GRAPH_POINTS", 500.0)

# Invalid override, should use default of 500
# 1 hour range with 500 points -> step = 3600/500 = 7.2
result = adjust_step_for_max_points(
"2024-01-01T00:00:00Z",
"2024-01-01T01:00:00Z",
step=None,
max_points_override=0,
)
assert result == 7.2

def test_override_with_explicit_step_adjusts_if_needed(self, monkeypatch):
"""When both step and override are provided, step is adjusted if it exceeds override."""
import holmes.plugins.toolsets.prometheus.prometheus as prom_module

monkeypatch.setattr(prom_module, "MAX_GRAPH_POINTS", 500.0)
monkeypatch.setattr(prom_module, "MAX_GRAPH_POINTS_HARD_LIMIT", 1000.0)

# 1 hour range, step=1 (would give 3600 points), max_points=1000
# 3600 > 1000, so adjusted_step = 3600/1000 = 3.6
result = adjust_step_for_max_points(
"2024-01-01T00:00:00Z",
"2024-01-01T01:00:00Z",
step=1,
max_points_override=1000,
)
assert result == 3.6
Loading