From 101ab57cddb8d691da2581331bae4c9f7ccd6576 Mon Sep 17 00:00:00 2001 From: Roi Glinik Date: Thu, 18 Dec 2025 13:03:07 +0200 Subject: [PATCH 01/27] ROB-2713 improve healthcheck for tempo and loki (#1189) Signed-off-by: Roi Glinik Signed-off-by: Filip Grebowski --- .../builtin-toolsets/coralogix-logs.md | 2 +- .../builtin-toolsets/grafanadashboards.md | 2 - .../builtin-toolsets/prometheus.md | 5 +- .../toolsets/grafana/base_grafana_toolset.py | 19 +++- holmes/plugins/toolsets/grafana/common.py | 1 - .../plugins/toolsets/grafana/grafana_api.py | 64 ----------- .../grafana/loki/toolset_grafana_loki.py | 25 ++++- .../toolsets/grafana/toolset_grafana.py | 11 +- .../toolsets/grafana/toolset_grafana_tempo.py | 23 ++-- .../plugins/toolsets/prometheus/prometheus.py | 7 +- loki/docker-compose.yaml | 1 + .../test_config_api_base_version.py | 1 - .../toolsets.yaml | 1 - tests/plugins/toolsets/grafana/conftest.py | 37 ++----- .../plugins/toolsets/grafana/test_grafana.py | 70 ++++++++++++ .../toolsets/grafana/test_grafana_api.py | 51 --------- .../toolsets/grafana/test_grafana_loki.py | 66 ------------ .../toolsets/grafana/test_grafana_tempo.py | 101 ++++++++++-------- .../grafana/test_grafana_tempo_tools.py | 42 -------- .../toolsets/prometheus/test_prometheus.py | 46 ++++++++ 20 files changed, 247 insertions(+), 328 deletions(-) delete mode 100644 holmes/plugins/toolsets/grafana/grafana_api.py create mode 100644 tests/plugins/toolsets/grafana/test_grafana.py delete mode 100644 tests/plugins/toolsets/grafana/test_grafana_api.py delete mode 100644 tests/plugins/toolsets/grafana/test_grafana_loki.py create mode 100644 tests/plugins/toolsets/prometheus/test_prometheus.py diff --git a/docs/data-sources/builtin-toolsets/coralogix-logs.md b/docs/data-sources/builtin-toolsets/coralogix-logs.md index 99ed8e23e3..2ede89ef79 100644 --- a/docs/data-sources/builtin-toolsets/coralogix-logs.md +++ b/docs/data-sources/builtin-toolsets/coralogix-logs.md @@ -29,7 +29,7 @@ toolsets: headers: Authorization: "Bearer " prometheus_url: "https://ng-api-http.eu2.coralogix.com/metrics" # replace domain - healthcheck: "api/v1/query?query=up" + ``` diff --git a/docs/data-sources/builtin-toolsets/grafanadashboards.md b/docs/data-sources/builtin-toolsets/grafanadashboards.md index 8749325caa..af4cfc2560 100644 --- a/docs/data-sources/builtin-toolsets/grafanadashboards.md +++ b/docs/data-sources/builtin-toolsets/grafanadashboards.md @@ -21,8 +21,6 @@ A [Grafana service account token](https://grafana.com/docs/grafana/latest/admini config: api_key: url: # e.g. https://acme-corp.grafana.net or http://localhost:3000 - # Optional: Custom health check endpoint (defaults to api/health) - # healthcheck: api/health # Optional: Additional headers for all requests # headers: # X-Custom-Header: "custom-value" diff --git a/docs/data-sources/builtin-toolsets/prometheus.md b/docs/data-sources/builtin-toolsets/prometheus.md index 5870bdbda4..8a171189e8 100644 --- a/docs/data-sources/builtin-toolsets/prometheus.md +++ b/docs/data-sources/builtin-toolsets/prometheus.md @@ -80,7 +80,7 @@ toolsets: enabled: true config: prometheus_url: http://:9090 - healthcheck: "-/healthy" # Path for health checking (default: -/healthy) + healthcheck: "/api/v1/query?query=up" # Default path for health checking. headers: Authorization: "Basic " @@ -105,7 +105,7 @@ toolsets: **Config option explanations:** - `prometheus_url`: The base URL for Prometheus. Should include protocol and port. -- `healthcheck`: Path used for health checking Prometheus or Mimir/Cortex endpoint. Defaults to `-/healthy` for Prometheus, use `/ready` for Grafana Mimir. +- `healthcheck`: Path used for health checking Prometheus or Mimir/Cortex endpoint. Defaults to `"/api/v1/query?query=up"`. - `headers`: Extra headers for all Prometheus HTTP requests (e.g., for authentication). - `default_metadata_time_window_hrs`: Time window (in hours) for metadata/discovery APIs to look for active metrics. Default: 1 hour. - `query_response_size_limit`: Maximum number of characters in a query response before truncation. Set to `null` to disable. Default: 20000. @@ -159,7 +159,6 @@ To use a Coralogix PromQL endpoint with HolmesGPT: prometheus/metrics: enabled: true config: - healthcheck: "/api/v1/query?query=up" # This is important for Coralogix prometheus_url: "https://prom-api.eu2.coralogix.com" # Use your region's endpoint headers: token: "{{ env.CORALOGIX_API_KEY }}" diff --git a/holmes/plugins/toolsets/grafana/base_grafana_toolset.py b/holmes/plugins/toolsets/grafana/base_grafana_toolset.py index af9e0a65c7..a89e75404b 100644 --- a/holmes/plugins/toolsets/grafana/base_grafana_toolset.py +++ b/holmes/plugins/toolsets/grafana/base_grafana_toolset.py @@ -1,12 +1,11 @@ import logging +from abc import abstractmethod from typing import Any, ClassVar, Tuple, Type from holmes.core.tools import CallablePrerequisite, Tool, Toolset, ToolsetTag from holmes.plugins.toolsets.consts import TOOLSET_CONFIG_MISSING_ERROR from holmes.plugins.toolsets.grafana.common import GrafanaConfig -from holmes.plugins.toolsets.grafana.grafana_api import grafana_health_check - class BaseGrafanaToolset(Toolset): config_class: ClassVar[Type[GrafanaConfig]] = GrafanaConfig @@ -39,12 +38,26 @@ def prerequisites_callable(self, config: dict[str, Any]) -> Tuple[bool, str]: try: self._grafana_config = self.config_class(**config) - return grafana_health_check(self._grafana_config) + return self.health_check() except Exception as e: logging.exception(f"Failed to set up grafana toolset {self.name}") return False, str(e) + @abstractmethod + def health_check(self) -> Tuple[bool, str]: + """ + Check if the toolset is healthy and can connect to its data source. + + Subclasses must implement this method to verify connectivity. + This method should NOT raise exceptions - catch them internally + and return (False, "error message") instead. + + Returns: + Tuple[bool, str]: (True, "") on success, (False, "error message") on failure. + """ + raise NotImplementedError("Subclasses must implement health_check()") + def get_example_config(self): example_config = GrafanaConfig( api_key="YOUR API KEY", diff --git a/holmes/plugins/toolsets/grafana/common.py b/holmes/plugins/toolsets/grafana/common.py index 185381970e..01c17ddbfe 100644 --- a/holmes/plugins/toolsets/grafana/common.py +++ b/holmes/plugins/toolsets/grafana/common.py @@ -19,7 +19,6 @@ class GrafanaConfig(BaseModel): url: str grafana_datasource_uid: Optional[str] = None external_url: Optional[str] = None - healthcheck: Optional[str] = "ready" def build_headers(api_key: Optional[str], additional_headers: Optional[Dict[str, str]]): diff --git a/holmes/plugins/toolsets/grafana/grafana_api.py b/holmes/plugins/toolsets/grafana/grafana_api.py deleted file mode 100644 index c3acc05d17..0000000000 --- a/holmes/plugins/toolsets/grafana/grafana_api.py +++ /dev/null @@ -1,64 +0,0 @@ -import logging -import requests # type: ignore -from typing import Tuple -import backoff - -from holmes.plugins.toolsets.grafana.common import ( - GrafanaConfig, - build_headers, -) - - -@backoff.on_exception( - backoff.expo, - requests.exceptions.RequestException, - max_tries=2, - giveup=lambda e: isinstance(e, requests.exceptions.HTTPError) - and e.response.status_code < 500, -) -def _try_health_url(url: str, headers: dict) -> None: - response = requests.get(url, headers=headers, timeout=5) - response.raise_for_status() - - -def grafana_health_check(config: GrafanaConfig) -> Tuple[bool, str]: - """ - Tests a healthcheck url for grafna loki. - 1. When using grafana as proxy, grafana_datasource_uid is provided, use the data source health url (docs are added). - 2. When using loki directly there are two cases. - a. Using loki cloud, health check is provided on the base url. - b. Using local loki, uses url/healthcheck default is url/ready - c. This function tries both direct loki cases for the user. - """ - health_urls = [] - if config.grafana_datasource_uid: - # https://grafana.com/docs/grafana/latest/developers/http_api/data_source/#check-data-source-health - health_urls.append( - f"{config.url}/api/datasources/uid/{config.grafana_datasource_uid}/health" - ) - else: - health_urls.append(f"{config.url}/{config.healthcheck}") - health_urls.append(config.url) # loki cloud uses no suffix. - g_headers = build_headers(api_key=config.api_key, additional_headers=config.headers) - - error_msg = "" - for url in health_urls: - try: - _try_health_url(url, g_headers) - return True, "" - except Exception as e: - logging.debug( - f"Failed to fetch grafana health status at {url}", exc_info=True - ) - error_msg += f"Failed to fetch grafana health status at {url}. {str(e)}\n" - - # Add helpful hint if this looks like a common misconfiguration - if config.grafana_datasource_uid and ":3100" in config.url: - error_msg += ( - "\n\nPossible configuration issue: grafana_datasource_uid is set but URL contains port 3100 " - "(typically used for direct Loki connections). Please verify:\n" - "- If connecting directly to Loki: remove grafana_datasource_uid from config\n" - "- If connecting via Grafana proxy: ensure URL points to Grafana (usually port 3000)" - ) - - return False, error_msg diff --git a/holmes/plugins/toolsets/grafana/loki/toolset_grafana_loki.py b/holmes/plugins/toolsets/grafana/loki/toolset_grafana_loki.py index 225c2bef50..a6a775dd9f 100644 --- a/holmes/plugins/toolsets/grafana/loki/toolset_grafana_loki.py +++ b/holmes/plugins/toolsets/grafana/loki/toolset_grafana_loki.py @@ -1,4 +1,4 @@ -from typing import Dict, Optional +from typing import Dict, Optional, Tuple import os import json from urllib.parse import quote @@ -67,6 +67,29 @@ def _build_grafana_loki_explore_url( class GrafanaLokiToolset(BaseGrafanaToolset): + def health_check(self) -> Tuple[bool, str]: + """Test a dummy query to check if service available.""" + (start, end) = process_timestamps_to_rfc3339( + start_timestamp=-1, + end_timestamp=None, + default_time_span_seconds=DEFAULT_TIME_SPAN_SECONDS, + ) + + c = self._grafana_config + try: + _ = execute_loki_query( + base_url=get_base_url(c), + api_key=c.api_key, + headers=c.headers, + query='{job="test_endpoint"}', + start=start, + end=end, + limit=1, + ) + except Exception as e: + return False, f"Unable to connect to Loki.\n{str(e)}" + return True, "" + def __init__(self): super().__init__( name="grafana/loki", diff --git a/holmes/plugins/toolsets/grafana/toolset_grafana.py b/holmes/plugins/toolsets/grafana/toolset_grafana.py index 3616bb044d..ba3813e8f3 100644 --- a/holmes/plugins/toolsets/grafana/toolset_grafana.py +++ b/holmes/plugins/toolsets/grafana/toolset_grafana.py @@ -1,5 +1,5 @@ import os -from typing import Any, ClassVar, Dict, Optional, Type, cast +from typing import Any, ClassVar, Dict, Optional, Type, cast, Tuple from urllib.parse import urlencode, urljoin from abc import ABC from holmes.core.tools import ( @@ -66,6 +66,15 @@ def __init__(self): os.path.dirname(__file__), "toolset_grafana_dashboard.jinja2" ) + def health_check(self) -> Tuple[bool, str]: + """Test connectivity by invoking GetDashboardTags tool.""" + tool = GetDashboardTags(self) + try: + _ = tool._make_grafana_request("/api/dashboards/tags", {}) + return True, "" + except Exception as e: + return False, f"Failed to connect to Grafana {str(e)}" + @property def grafana_config(self) -> GrafanaDashboardConfig: return cast(GrafanaDashboardConfig, self._grafana_config) diff --git a/holmes/plugins/toolsets/grafana/toolset_grafana_tempo.py b/holmes/plugins/toolsets/grafana/toolset_grafana_tempo.py index 536d572472..dc8b72f844 100644 --- a/holmes/plugins/toolsets/grafana/toolset_grafana_tempo.py +++ b/holmes/plugins/toolsets/grafana/toolset_grafana_tempo.py @@ -148,22 +148,17 @@ def get_example_config(self): def grafana_config(self) -> GrafanaTempoConfig: return cast(GrafanaTempoConfig, self._grafana_config) - def prerequisites_callable(self, config: dict[str, Any]) -> Tuple[bool, str]: - """Check Tempo connectivity using the echo endpoint.""" - # First call parent to validate config - success, msg = super().prerequisites_callable(config) - if not success: - return success, msg - - # Then check Tempo-specific echo endpoint + def health_check(self) -> Tuple[bool, str]: + """Test a dummy query to check if service available.""" try: - api = GrafanaTempoAPI(self.grafana_config) - if api.query_echo_endpoint(): - return True, "Successfully connected to Tempo" - else: - return False, "Failed to connect to Tempo echo endpoint" + _ = GrafanaTempoAPI(self.grafana_config).search_traces_by_query( + q='{ .service.name = "test-endpoint" }', + limit=1, + ) except Exception as e: - return False, f"Failed to connect to Tempo: {str(e)}" + return False, f"Unable to connect to Tempo.\n{str(e)}" + + return True, "" def build_k8s_filters( self, params: Dict[str, Any], use_exact_match: bool diff --git a/holmes/plugins/toolsets/prometheus/prometheus.py b/holmes/plugins/toolsets/prometheus/prometheus.py index 300e975330..9ea25bd2ac 100644 --- a/holmes/plugins/toolsets/prometheus/prometheus.py +++ b/holmes/plugins/toolsets/prometheus/prometheus.py @@ -54,7 +54,6 @@ class PrometheusConfig(BaseModel): # URL is optional because it can be set with an env var prometheus_url: Optional[str] - healthcheck: str = "-/healthy" # New config for default time window for metadata APIs default_metadata_time_window_hrs: int = DEFAULT_METADATA_TIME_WINDOW_HRS # Default: only show metrics active in the last hour @@ -128,9 +127,6 @@ def validate_prom_config(self): ) # If openshift is enabled, and the user didn't configure auth headers, we will try to load the token from the service account. if IS_OPENSHIFT: - if self.healthcheck == "-/healthy": - self.healthcheck = "api/v1/query?query=up" - if self.headers.get("Authorization"): return self @@ -150,7 +146,6 @@ class AMPConfig(PrometheusConfig): aws_secret_access_key: Optional[str] = None aws_region: str aws_service_name: str = "aps" - healthcheck: str = "api/v1/query?query=up" prometheus_ssl_enabled: bool = False assume_role_arn: Optional[str] = None @@ -1584,7 +1579,7 @@ def _is_healthy(self) -> Tuple[bool, str]: f"Toolset {self.name} failed to initialize because prometheus is not configured correctly", ) - url = urljoin(self.config.prometheus_url, self.config.healthcheck) + url = urljoin(self.config.prometheus_url, "api/v1/query?query=up") try: response = do_request( config=self.config, diff --git a/loki/docker-compose.yaml b/loki/docker-compose.yaml index 1201e2038b..8fcd01a6c0 100644 --- a/loki/docker-compose.yaml +++ b/loki/docker-compose.yaml @@ -33,6 +33,7 @@ services: datasources: - name: Loki type: loki + uid: loki-test-uid access: proxy orgId: 1 url: http://loki:3100 diff --git a/tests/config_class/test_config_api_base_version.py b/tests/config_class/test_config_api_base_version.py index 9a8416fd38..f0dc79ccd4 100644 --- a/tests/config_class/test_config_api_base_version.py +++ b/tests/config_class/test_config_api_base_version.py @@ -2,7 +2,6 @@ from pydantic import SecretStr import yaml - from holmes.config import Config diff --git a/tests/llm/fixtures/test_ask_holmes/175_coralogix_metrics_frontend/toolsets.yaml b/tests/llm/fixtures/test_ask_holmes/175_coralogix_metrics_frontend/toolsets.yaml index 12460267ea..6cacd00971 100644 --- a/tests/llm/fixtures/test_ask_holmes/175_coralogix_metrics_frontend/toolsets.yaml +++ b/tests/llm/fixtures/test_ask_holmes/175_coralogix_metrics_frontend/toolsets.yaml @@ -6,7 +6,6 @@ toolsets: prometheus_url: "https://ng-api-http.{{env.CORALOGIX_DOMAIN}}/metrics" headers: Authorization: "Bearer {{env.CORALOGIX_API_KEY}}" - healthcheck: "api/v1/query?query=up" kubernetes/logs: enabled: false kubernetes/core: diff --git a/tests/plugins/toolsets/grafana/conftest.py b/tests/plugins/toolsets/grafana/conftest.py index 8ba37e7d7d..a1eb289774 100644 --- a/tests/plugins/toolsets/grafana/conftest.py +++ b/tests/plugins/toolsets/grafana/conftest.py @@ -1,30 +1,13 @@ -import os -import requests # type: ignore +import socket +from typing import Optional -def check_grafana_connectivity(): - """Check if required Grafana environment variables are set and server is reachable""" - REQUIRED_ENV_VARS = [ - "GRAFANA_URL", - "GRAFANA_API_KEY", - ] - - missing_vars = [var for var in REQUIRED_ENV_VARS if os.environ.get(var) is None] - - if missing_vars: - return f"{', '.join(missing_vars)} must be set" - - # Check if Grafana server is reachable +def check_service_running( + service_name: str, port: int, host: str = "localhost", timeout: float = 2.0 +) -> Optional[str]: + """Check if a service is running and return skip reason if not.""" try: - GRAFANA_URL = os.environ.get("GRAFANA_URL", "") - GRAFANA_API_KEY = os.environ.get("GRAFANA_API_KEY") - headers = {} - if GRAFANA_API_KEY: - headers["Authorization"] = f"Bearer {GRAFANA_API_KEY}" - response = requests.get(f"{GRAFANA_URL}/ready", headers=headers, timeout=2) - if response.status_code != 200: - return f"Grafana server not reachable at {GRAFANA_URL}/ready (status: {response.status_code}). Set GRAFANA_URL and GRAFANA_API_KEY to run Grafana tests" - except Exception: - return f"Grafana server not reachable at {GRAFANA_URL}/ready. Set GRAFANA_URL and GRAFANA_API_KEY to run Grafana tests" - - return None + with socket.create_connection((host, port), timeout=timeout): + return None + except (socket.timeout, ConnectionRefusedError, OSError): + return f"{service_name} is not running on {host}:{port}" diff --git a/tests/plugins/toolsets/grafana/test_grafana.py b/tests/plugins/toolsets/grafana/test_grafana.py new file mode 100644 index 0000000000..4ecf4677bd --- /dev/null +++ b/tests/plugins/toolsets/grafana/test_grafana.py @@ -0,0 +1,70 @@ +import pytest + +from holmes.core.tools import ToolsetStatusEnum +from holmes.plugins.toolsets.grafana.loki.toolset_grafana_loki import ( + GrafanaLokiToolset, +) +from holmes.plugins.toolsets.grafana.toolset_grafana import GrafanaToolset +from tests.plugins.toolsets.grafana.conftest import check_service_running + +# Skip all tests in this module if Grafana and loki are not running. use loki/docker-compose.yaml +skip_reason = check_service_running("Grafana", 3000) +if skip_reason: + pytestmark = pytest.mark.skip(reason=skip_reason) + + +def test_grafana_toolset_direct_health_check(): + toolset = GrafanaToolset() + toolset.config = {"url": "http://localhost:3000/"} + toolset.check_prerequisites() + + assert toolset.error is None + assert toolset.status == ToolsetStatusEnum.ENABLED + + +def test_grafana_toolset_error_health_check(): + toolset = GrafanaToolset() + toolset.config = {"url": "http://localhost:2000/"} + toolset.check_prerequisites() + + assert ( + "Failed to connect to Grafana HTTPConnectionPool(host='localhost', port=2000): Max retries exceeded with url: /api/dashboards/tags" + in toolset.error + ) + assert toolset.status == ToolsetStatusEnum.FAILED + + +def test_loki_toolset_direct_health_check(): + toolset = GrafanaLokiToolset() + toolset.config = {"url": "http://localhost:3100/"} + toolset.check_prerequisites() + + assert toolset.error is None + assert toolset.status == ToolsetStatusEnum.ENABLED + + +def test_loki_datasource_toolset_health_check(): + toolset = GrafanaLokiToolset() + toolset.config = { + "url": "http://localhost:3000/", + "grafana_datasource_uid": "loki-test-uid", + } + toolset.check_prerequisites() + + assert toolset.error is None + assert toolset.status == ToolsetStatusEnum.ENABLED + + +def test_loki_datasource_toolset_error_health_check(): + toolset = GrafanaLokiToolset() + toolset.config = { + "url": "http://localhost:3000/", + "grafana_datasource_uid": "wrong-uid", + } + toolset.check_prerequisites() + + assert ( + "Unable to connect to Loki.\nFailed to query Loki logs: 404 Client Error: Not Found for url: http://localhost:3000//api/datasources/proxy/uid/wrong-uid/loki/api/v1/query_range" + in toolset.error + ) + assert toolset.status == ToolsetStatusEnum.FAILED diff --git a/tests/plugins/toolsets/grafana/test_grafana_api.py b/tests/plugins/toolsets/grafana/test_grafana_api.py deleted file mode 100644 index 2ef8632097..0000000000 --- a/tests/plugins/toolsets/grafana/test_grafana_api.py +++ /dev/null @@ -1,51 +0,0 @@ -import pytest -from unittest.mock import Mock, patch -import requests # type: ignore - -from holmes.plugins.toolsets.grafana.common import GrafanaConfig -from holmes.plugins.toolsets.grafana.grafana_api import grafana_health_check - - -@pytest.fixture -def mock_requests_get(): - """Fixture to mock requests.get""" - with patch("holmes.plugins.toolsets.grafana.grafana_api.requests.get") as mock: - yield mock - - -class TestGrafanaHealthCheck: - """Test cases for grafana_health_check function""" - - def test_first_url_succeeds(self, mock_requests_get): - """Test that function returns True when first URL succeeds""" - config = GrafanaConfig( - url="http://grafana:3000", - grafana_datasource_uid="loki-uid", - ) - - mock_response = Mock() - mock_response.raise_for_status = Mock() - mock_requests_get.return_value = mock_response - - success, error = grafana_health_check(config) - - assert success is True - assert error == "" - assert mock_requests_get.call_count == 1 - - def test_second_url_succeeds_after_first_fails(self, mock_requests_get): - """Test that function tries second URL when first fails""" - config = GrafanaConfig( - url="http://grafana:3000", - ) - - mock_requests_get.side_effect = [ - requests.exceptions.ConnectionError("Connection failed"), - requests.exceptions.ConnectionError("Connection failed"), - Mock(raise_for_status=Mock()), - ] - - success, error = grafana_health_check(config) - assert success is True - assert error == "" - assert mock_requests_get.call_count > 2 diff --git a/tests/plugins/toolsets/grafana/test_grafana_loki.py b/tests/plugins/toolsets/grafana/test_grafana_loki.py deleted file mode 100644 index 599527071a..0000000000 --- a/tests/plugins/toolsets/grafana/test_grafana_loki.py +++ /dev/null @@ -1,66 +0,0 @@ -import os -from typing import Any -from holmes.core.tools import ToolsetStatusEnum -from holmes.plugins.toolsets.grafana.grafana_api import grafana_health_check -import pytest - -from holmes.plugins.toolsets.grafana.loki.toolset_grafana_loki import ( - GrafanaLokiToolset, -) -from tests.plugins.toolsets.grafana.conftest import check_grafana_connectivity -from holmes.plugins.toolsets.grafana.common import GrafanaConfig - -# Use pytest.mark.skip (not skipif) to show a single grouped skip line for the entire module -# Will show: "SKIPPED [4] module.py: reason" instead of 4 separate skip lines -skip_reason = check_grafana_connectivity() -if skip_reason: - pytestmark = pytest.mark.skip(reason=skip_reason) - -TEST_NAMESPACE = "default" -TEST_POD_NAME = "robusta-holmes-7cd886dc86-x5zfd" -TEST_SEARCH_TERM = "WARNING" - -# the date range below combined with the search term is expected to return a single log line -TEST_START_TIME = "2025-05-20T05:11:00Z" -TEST_END_TIME = "2025-05-20T05:12:00Z" - - -GRAFANA_API_KEY = os.environ.get("GRAFANA_API_KEY") -GRAFANA_URL = os.environ.get("GRAFANA_URL", "") -GRAFANA_LOKI_DATASOURCE_UID = os.environ.get("GRAFANA_LOKI_DATASOURCE_UID") -GRAFANA_LOKI_X_SCOPE_ORGID = os.environ.get("GRAFANA_LOKI_X_SCOPE_ORGID") - - -@pytest.fixture -def loki_config() -> GrafanaConfig: - # All checks done at module level - env vars and connectivity guaranteed - config_dict: dict[str, Any] = { - "api_key": GRAFANA_API_KEY, - "url": GRAFANA_URL, - "grafana_datasource_uid": GRAFANA_LOKI_DATASOURCE_UID, - } - - if GRAFANA_LOKI_X_SCOPE_ORGID: - config_dict["headers"] = {"X-Scope-OrgID": GRAFANA_LOKI_X_SCOPE_ORGID} - - return GrafanaConfig(**config_dict) - - -@pytest.fixture -def loki_toolset(loki_config) -> GrafanaLokiToolset: - """Create an OpenSearchLogsToolset with the test configuration""" - toolset = GrafanaLokiToolset() - toolset.config = loki_config.model_dump() - toolset.check_prerequisites() - - assert toolset.error is None - assert toolset.status == ToolsetStatusEnum.ENABLED - - return toolset - - -def test_grafana_loki_health_check(loki_config): - success, error_message = grafana_health_check(loki_config) - - assert not error_message - assert success diff --git a/tests/plugins/toolsets/grafana/test_grafana_tempo.py b/tests/plugins/toolsets/grafana/test_grafana_tempo.py index ca122da451..981be08f2b 100644 --- a/tests/plugins/toolsets/grafana/test_grafana_tempo.py +++ b/tests/plugins/toolsets/grafana/test_grafana_tempo.py @@ -1,20 +1,21 @@ import json import os -from holmes.plugins.toolsets.grafana.grafana_api import grafana_health_check + import pytest +import requests # type: ignore +import responses -from holmes.plugins.toolsets.grafana.common import GrafanaTempoConfig from holmes.plugins.toolsets.grafana.trace_parser import process_trace -from tests.plugins.toolsets.grafana.conftest import check_grafana_connectivity +from tests.plugins.toolsets.grafana.conftest import check_service_running -GRAFANA_API_KEY = os.environ.get("GRAFANA_API_KEY", "") -GRAFANA_URL = os.environ.get("GRAFANA_URL", "") -GRAFANA_TEMPO_DATASOURCE_UID = os.environ.get("GRAFANA_TEMPO_DATASOURCE_UID", "") +from holmes.core.tools import ToolsetStatusEnum +from holmes.plugins.toolsets.grafana.toolset_grafana_tempo import ( + GrafanaTempoToolset, +) -# Use pytest.mark.skip (not skipif) to show a single grouped skip line for the entire module -# Will show: "SKIPPED [4] module.py: reason" instead of 4 separate skip lines -skip_reason = check_grafana_connectivity() +# use docker compose setup from https://github.com/grafana/tempo/blob/main/example/docker-compose/local/readme.md to run local grafana and tempo. +skip_reason = check_service_running("Grafana", 3000) if skip_reason: pytestmark = pytest.mark.skip(reason=skip_reason) @@ -53,48 +54,60 @@ def test_process_trace_json(): assert result.strip() == expected_result.strip() -# def test_grafana_tempo_has_prompt(): -# toolset = GrafanaTempoToolset() -# tool = GetTempoTraces(toolset) -# assert tool.name is not None -# assert toolset.llm_instructions is not None -# assert tool.name in toolset.llm_instructions - +def test_tempo_toolset_direct_health_check(): + toolset = GrafanaTempoToolset() + toolset.config = {"url": "http://localhost:3200/"} + toolset.check_prerequisites() -# def test_grafana_query_loki_logs_by_pod(): -# config = { -# "api_key": GRAFANA_API_KEY, -# "headers": {}, -# "url": GRAFANA_URL, -# "grafana_datasource_uid": GRAFANA_TEMPO_DATASOURCE_UID, -# } + assert toolset.error is None + assert toolset.status == ToolsetStatusEnum.ENABLED -# if not GRAFANA_TEMPO_DATASOURCE_UID: -# config["headers"]["X-Scope-OrgID"] = ( -# "1" # standalone tempo likely requires an orgid -# ) -# toolset = GrafanaTempoToolset() -# toolset.config = config -# toolset.check_prerequisites() +def test_tempo_datasource_toolset_health_check(): + toolset = GrafanaTempoToolset() + toolset.config = { + "url": "http://localhost:3000/", + "grafana_datasource_uid": "tempo-streaming-enabled", + } + toolset.check_prerequisites() -# assert toolset.error is None -# assert toolset.status == ToolsetStatusEnum.ENABLED + assert toolset.error is None + assert toolset.status == ToolsetStatusEnum.ENABLED -# tool = GetTempoTraces(toolset) -# # just tests that this does not throw -# tool.invoke(params={"min_duration": "5"}) +def test_tempo_datasource_toolset_wrong_url_health_check(): + toolset = GrafanaTempoToolset() + toolset.config = { + "url": "http://localhost:2000/", + "grafana_datasource_uid": "tempo-streaming-enabled", + } + toolset.check_prerequisites() -def test_grafana_loki_health_check(): - config = GrafanaTempoConfig( - api_key=GRAFANA_API_KEY, - headers=None, - url=GRAFANA_URL, - grafana_datasource_uid=GRAFANA_TEMPO_DATASOURCE_UID, + assert ( + "Unable to connect to Tempo.\nHTTPConnectionPool(host='localhost', port=2000): Max retries exceeded with url: /api/datasources/proxy/uid/tempo-streaming-enabled/api/search" + in toolset.error ) + assert toolset.status == ToolsetStatusEnum.FAILED + + +def test_tempo_datasource_toolset_health_check_exceptions(): + """Test that health check handles request exceptions properly with backoff retries.""" + toolset = GrafanaTempoToolset() + toolset.config = { + "url": "http://localhost:3000/", + "grafana_datasource_uid": "tempo-streaming-enabled", + } + + with responses.RequestsMock() as rsps: + rsps.add( + responses.GET, + 'http://localhost:3000//api/datasources/proxy/uid/tempo-streaming-enabled/api/search?q={ .service.name = "test-endpoint" }&limit=1', + body=requests.exceptions.ConnectionError("Connection refused"), + status=400, + ) - success, error_message = grafana_health_check(config) + toolset.check_prerequisites() - assert not error_message - assert success + assert len(rsps.calls) == 3, "Expected 3 retries due to backoff" + assert toolset.status == ToolsetStatusEnum.FAILED + assert "Connection refused" in toolset.error diff --git a/tests/plugins/toolsets/grafana/test_grafana_tempo_tools.py b/tests/plugins/toolsets/grafana/test_grafana_tempo_tools.py index 16942c73bd..1ede3048f0 100644 --- a/tests/plugins/toolsets/grafana/test_grafana_tempo_tools.py +++ b/tests/plugins/toolsets/grafana/test_grafana_tempo_tools.py @@ -707,45 +707,3 @@ def test_toolset_has_all_tools(self): for expected in expected_tools: assert expected in tool_names - - def test_prerequisites_success(self, tempo_config): - """Test successful prerequisites check.""" - toolset = GrafanaTempoToolset() - - config_dict = tempo_config.model_dump() - - with patch( - "holmes.plugins.toolsets.grafana.base_grafana_toolset.grafana_health_check" - ) as mock_health: - mock_health.return_value = (True, "Success") - - with patch( - "holmes.plugins.toolsets.grafana.grafana_tempo_api.GrafanaTempoAPI.query_echo_endpoint" - ) as mock_echo: - mock_echo.return_value = True - - success, message = toolset.prerequisites_callable(config_dict) - - assert success is True - assert "Successfully connected to Tempo" in message - - def test_prerequisites_echo_failure(self, tempo_config): - """Test prerequisites failure on echo endpoint.""" - toolset = GrafanaTempoToolset() - - config_dict = tempo_config.model_dump() - - with patch( - "holmes.plugins.toolsets.grafana.base_grafana_toolset.grafana_health_check" - ) as mock_health: - mock_health.return_value = (True, "Success") - - with patch( - "holmes.plugins.toolsets.grafana.grafana_tempo_api.GrafanaTempoAPI.query_echo_endpoint" - ) as mock_echo: - mock_echo.return_value = False - - success, message = toolset.prerequisites_callable(config_dict) - - assert success is False - assert "Failed to connect to Tempo echo endpoint" in message diff --git a/tests/plugins/toolsets/prometheus/test_prometheus.py b/tests/plugins/toolsets/prometheus/test_prometheus.py new file mode 100644 index 0000000000..0afe523a5c --- /dev/null +++ b/tests/plugins/toolsets/prometheus/test_prometheus.py @@ -0,0 +1,46 @@ +from holmes.core.tools import ToolsetStatusEnum +from holmes.plugins.toolsets.prometheus.prometheus import PrometheusToolset +from tests.plugins.toolsets.grafana.conftest import check_service_running +import pytest + +skip_reason = check_service_running("Grafana", 9000) +if skip_reason: + pytestmark = pytest.mark.skip(reason=skip_reason) + + +# Use docker compose with https://github.com/grafana/mimir/blob/main/docs/sources/mimir/get-started/play-with-grafana-mimir/index.md +def test_mimir_datasource_toolset_health_check(): + toolset = PrometheusToolset() + toolset.config = { + "prometheus_url": "http://localhost:9000/api/datasources/proxy/uid/PAE45454D0EDB9216", + } + toolset.check_prerequisites() + + assert toolset.error is None + assert toolset.status == ToolsetStatusEnum.ENABLED + + +def test_mimir_datasource_toolset_bad_uid_health_check(): + toolset = PrometheusToolset() + toolset.config = { + "prometheus_url": "http://localhost:9000/api/datasources/proxy/uid/PAE45454D0EDB9216111", + } + toolset.check_prerequisites() + + assert ( + "Failed to connect to Prometheus at http://localhost:9000/api/datasources/proxy/uid/PAE45454D0EDB9216111/api/v1/query?query=up: HTTP 404" + in toolset.error + ) + assert toolset.status == ToolsetStatusEnum.FAILED + + +def test_mimir_direct_toolset_health_check(): + toolset = PrometheusToolset() + toolset.config = { + "prometheus_url": "http://localhost:9009/prometheus", + "headers": {"X-Scope-OrgID": "DEMO"}, + } + toolset.check_prerequisites() + + assert toolset.error is None + assert toolset.status == ToolsetStatusEnum.ENABLED From 75f4f7b19a188d79521dd1b21e2ed6d5c3ec2cf9 Mon Sep 17 00:00:00 2001 From: Filip Grebowski Date: Fri, 19 Dec 2025 13:39:39 +0000 Subject: [PATCH 02/27] Adding Backstage Signed-off-by: Filip Grebowski --- docs/data-sources/builtin-toolsets/.nav.yml | 67 ++++++++++--------- .../builtin-toolsets/backstage.md | 21 ++++++ docs/data-sources/builtin-toolsets/index.md | 1 + 3 files changed, 56 insertions(+), 33 deletions(-) create mode 100644 docs/data-sources/builtin-toolsets/backstage.md diff --git a/docs/data-sources/builtin-toolsets/.nav.yml b/docs/data-sources/builtin-toolsets/.nav.yml index f126eb2bec..fadef796c6 100644 --- a/docs/data-sources/builtin-toolsets/.nav.yml +++ b/docs/data-sources/builtin-toolsets/.nav.yml @@ -1,34 +1,35 @@ nav: - - index.md - - AKS Node Health: aks-node-health.md - - ArgoCD: argocd.md - - AWS (MCP): aws.md - - Azure Kubernetes Service: aks.md - - Azure SQL Database: azure-sql.md - - Bash: bash.md - - Cilium: cilium.md - - Confluence: confluence.md - - Coralogix: coralogix-logs.md - - DataDog: datadog.md - - Datetime: datetime.md - - Docker: docker.md - - GitHub: github.md - - Grafana Dashboards: grafanadashboards.md - - Loki: grafanaloki.md - - Tempo: grafanatempo.md - - Helm: helm.md - - Internet: internet.md - - Kafka: kafka.md - - Kubernetes: kubernetes.md - - MariaDB (MCP): mariadb-mcp.md - - MongoDB Atlas: mongodb-atlas.md - - New Relic: newrelic.md - - Notion: notion.md - - OpenSearch logs: opensearch-logs.md - - OpenSearch status: opensearch-status.md - - OpenShift: openshift.md - - Prometheus: prometheus.md - - RabbitMQ: rabbitmq.md - - Robusta: robusta.md - - ServiceNow: servicenow.md - - Slab: slab.md + - index.md + - AKS Node Health: aks-node-health.md + - ArgoCD: argocd.md + - AWS (MCP): aws.md + - Azure Kubernetes Service: aks.md + - Azure SQL Database: azure-sql.md + - Backstage: backstage.md + - Bash: bash.md + - Cilium: cilium.md + - Confluence: confluence.md + - Coralogix: coralogix-logs.md + - DataDog: datadog.md + - Datetime: datetime.md + - Docker: docker.md + - GitHub: github.md + - Grafana Dashboards: grafanadashboards.md + - Loki: grafanaloki.md + - Tempo: grafanatempo.md + - Helm: helm.md + - Internet: internet.md + - Kafka: kafka.md + - Kubernetes: kubernetes.md + - MariaDB (MCP): mariadb-mcp.md + - MongoDB Atlas: mongodb-atlas.md + - New Relic: newrelic.md + - Notion: notion.md + - OpenSearch logs: opensearch-logs.md + - OpenSearch status: opensearch-status.md + - OpenShift: openshift.md + - Prometheus: prometheus.md + - RabbitMQ: rabbitmq.md + - Robusta: robusta.md + - ServiceNow: servicenow.md + - Slab: slab.md diff --git a/docs/data-sources/builtin-toolsets/backstage.md b/docs/data-sources/builtin-toolsets/backstage.md new file mode 100644 index 0000000000..3d3206d613 --- /dev/null +++ b/docs/data-sources/builtin-toolsets/backstage.md @@ -0,0 +1,21 @@ +# Backstage + +Backstage gives engineering organizations a central developer portal – a single catalog where every software component, API, environment, and deployment lives. + +Each entity in the catalog carries rich metadata: documentation, CI/CD pipeline data, Kubernetes deployment status, runtime metrics, APIs, dependencies, and more. + +HolmesGPT taps into that information. Instead of jumping between dashboards and consoles, you can simply ask questions about the component you’re looking at – and Holmes runs an investigation for you. + +Below is a short walkthrough of the HolmesGPT–Backstage integration. It demonstrates how to start an investigation, inspect the investigation plan, and understand why a service may be failing, including reviewing the root cause analysis generated by HolmesGPT. + +
+ +
+ +If you’re already running Backstage and want to bring AI-driven diagnostics directly into your developer portal, reach out to us via +[email](arik@robusta.dev) or [slack](robustacommunity.slack.com). You can also use the [this form](https://robusta-dev.typeform.com/to/xhyQYgLx) to register your interest. diff --git a/docs/data-sources/builtin-toolsets/index.md b/docs/data-sources/builtin-toolsets/index.md index aeb243e045..3d6758152a 100644 --- a/docs/data-sources/builtin-toolsets/index.md +++ b/docs/data-sources/builtin-toolsets/index.md @@ -11,6 +11,7 @@ HolmesGPT includes pre-built integrations for popular monitoring and observabili - [:material-aws:{ .lg .middle } **AWS**](aws.md) - [:material-microsoft-azure:{ .lg .middle } **Azure Kubernetes Service**](aks.md) - [:material-database:{ .lg .middle } **Azure SQL Database**](azure-sql.md) +- [:material-alpha-b-circle:{ .lg .middle } **Backstage**](backstage.md) - [:material-console:{ .lg .middle } **Bash**](bash.md) - [:simple-cilium:{ .lg .middle } **Cilium**](cilium.md) - [:simple-confluence:{ .lg .middle } **Confluence**](confluence.md) From b4db929497da78a70fd134024a0e551571ab0636 Mon Sep 17 00:00:00 2001 From: Filip Grebowski Date: Fri, 19 Dec 2025 14:15:23 +0000 Subject: [PATCH 03/27] Addressed review bot comments Signed-off-by: Filip Grebowski --- docs/data-sources/builtin-toolsets/backstage.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/data-sources/builtin-toolsets/backstage.md b/docs/data-sources/builtin-toolsets/backstage.md index 3d3206d613..5144ce8b50 100644 --- a/docs/data-sources/builtin-toolsets/backstage.md +++ b/docs/data-sources/builtin-toolsets/backstage.md @@ -18,4 +18,4 @@ Below is a short walkthrough of the HolmesGPT–Backstage integration. It demons If you’re already running Backstage and want to bring AI-driven diagnostics directly into your developer portal, reach out to us via -[email](arik@robusta.dev) or [slack](robustacommunity.slack.com). You can also use the [this form](https://robusta-dev.typeform.com/to/xhyQYgLx) to register your interest. +[email](arik@robusta.dev) or [slack](robustacommunity.slack.com). You can also use [this form](https://robusta-dev.typeform.com/to/xhyQYgLx) to register your interest. From 8e909afc83002bb75c88a600053428632d433af7 Mon Sep 17 00:00:00 2001 From: Filip Grebowski Date: Sat, 27 Dec 2025 13:08:38 +0000 Subject: [PATCH 04/27] Moved Backstage to Installation and separated 3rd party intregrations to separate pages Signed-off-by: Filip Grebowski --- docs/data-sources/builtin-toolsets/.nav.yml | 1 - .../builtin-toolsets/backstage.md | 21 --- docs/data-sources/builtin-toolsets/index.md | 1 - docs/installation/.nav.yml | 11 +- docs/installation/backstage-installation.md | 31 ++++ docs/installation/cli-installation.md | 26 +-- docs/installation/k9s-installation.md | 138 ++++++++++++++++ docs/installation/slack-installation.md | 19 +++ docs/installation/ui-installation.md | 148 +----------------- mkdocs.yml | 62 ++++++++ 10 files changed, 273 insertions(+), 185 deletions(-) delete mode 100644 docs/data-sources/builtin-toolsets/backstage.md create mode 100644 docs/installation/backstage-installation.md create mode 100644 docs/installation/k9s-installation.md create mode 100644 docs/installation/slack-installation.md diff --git a/docs/data-sources/builtin-toolsets/.nav.yml b/docs/data-sources/builtin-toolsets/.nav.yml index fadef796c6..b9ef90f777 100644 --- a/docs/data-sources/builtin-toolsets/.nav.yml +++ b/docs/data-sources/builtin-toolsets/.nav.yml @@ -5,7 +5,6 @@ nav: - AWS (MCP): aws.md - Azure Kubernetes Service: aks.md - Azure SQL Database: azure-sql.md - - Backstage: backstage.md - Bash: bash.md - Cilium: cilium.md - Confluence: confluence.md diff --git a/docs/data-sources/builtin-toolsets/backstage.md b/docs/data-sources/builtin-toolsets/backstage.md deleted file mode 100644 index 5144ce8b50..0000000000 --- a/docs/data-sources/builtin-toolsets/backstage.md +++ /dev/null @@ -1,21 +0,0 @@ -# Backstage - -Backstage gives engineering organizations a central developer portal – a single catalog where every software component, API, environment, and deployment lives. - -Each entity in the catalog carries rich metadata: documentation, CI/CD pipeline data, Kubernetes deployment status, runtime metrics, APIs, dependencies, and more. - -HolmesGPT taps into that information. Instead of jumping between dashboards and consoles, you can simply ask questions about the component you’re looking at – and Holmes runs an investigation for you. - -Below is a short walkthrough of the HolmesGPT–Backstage integration. It demonstrates how to start an investigation, inspect the investigation plan, and understand why a service may be failing, including reviewing the root cause analysis generated by HolmesGPT. - -
- -
- -If you’re already running Backstage and want to bring AI-driven diagnostics directly into your developer portal, reach out to us via -[email](arik@robusta.dev) or [slack](robustacommunity.slack.com). You can also use [this form](https://robusta-dev.typeform.com/to/xhyQYgLx) to register your interest. diff --git a/docs/data-sources/builtin-toolsets/index.md b/docs/data-sources/builtin-toolsets/index.md index 3d6758152a..aeb243e045 100644 --- a/docs/data-sources/builtin-toolsets/index.md +++ b/docs/data-sources/builtin-toolsets/index.md @@ -11,7 +11,6 @@ HolmesGPT includes pre-built integrations for popular monitoring and observabili - [:material-aws:{ .lg .middle } **AWS**](aws.md) - [:material-microsoft-azure:{ .lg .middle } **Azure Kubernetes Service**](aks.md) - [:material-database:{ .lg .middle } **Azure SQL Database**](azure-sql.md) -- [:material-alpha-b-circle:{ .lg .middle } **Backstage**](backstage.md) - [:material-console:{ .lg .middle } **Bash**](bash.md) - [:simple-cilium:{ .lg .middle } **Cilium**](cilium.md) - [:simple-confluence:{ .lg .middle } **Confluence**](confluence.md) diff --git a/docs/installation/.nav.yml b/docs/installation/.nav.yml index ba14fa7162..2a7334e488 100644 --- a/docs/installation/.nav.yml +++ b/docs/installation/.nav.yml @@ -1,5 +1,8 @@ nav: - - Install CLI: cli-installation.md - - Install UI/Slack/K9s: ui-installation.md - - Install Helm Chart: kubernetes-installation.md - - Install Python SDK: python-installation.md + - Install CLI: cli-installation.md + - Install UI (3rd party): ui-installation.md + - Install Slack Bot (3rd party): slack-installation.md + - Install Backstage (3rd party): backstage-installation.md + - Install K9s: k9s-installation.md + - Install Helm Chart: kubernetes-installation.md + - Install Python SDK: python-installation.md diff --git a/docs/installation/backstage-installation.md b/docs/installation/backstage-installation.md new file mode 100644 index 0000000000..4a0064f40f --- /dev/null +++ b/docs/installation/backstage-installation.md @@ -0,0 +1,31 @@ +# Install Backstage (third party) + +There’s now a third-party integration that brings **HolmesGPT** by **Robusta.dev** directly into Backstage. The integration is currently available in **closed beta**. + +Instead of jumping between dashboards, logs, and consoles, you can ask questions directly from the Backstage component you’re already looking at. HolmesGPT then runs an automated investigation on your behalf. + +Because Backstage entities already contain rich metadata, HolmesGPT can tap into: + +- Service documentation +- CI/CD pipeline data +- Kubernetes deployment status +- Runtime metrics +- APIs and dependencies +- And more + +All of this context is used to build a targeted investigation tailored to the specific service. + +Below is a short walkthrough of the HolmesGPT–Backstage integration. It demonstrates how to start an investigation, inspect the investigation plan, and understand why a service may be failing, including reviewing the root cause analysis generated by HolmesGPT. + +
+ +
+ +If you’re already running Backstage and want to bring AI-driven diagnostics into your developer portal, you can request access to the closed beta by contacting the Robusta.dev team via [email](mailto:arik@robusta.dev) or [slack](https://robustacommunity.slack.com). You can also use [this form](https://robusta-dev.typeform.com/to/xhyQYgLx) to register your interest. + + diff --git a/docs/installation/cli-installation.md b/docs/installation/cli-installation.md index fad3ba96ac..f818c06ccb 100644 --- a/docs/installation/cli-installation.md +++ b/docs/installation/cli-installation.md @@ -259,21 +259,21 @@ You can define multiple models in a YAML file and reference them by name in the ```yaml # model_list.yaml sonnet: - aws_access_key_id: 'your-access-key' - aws_region_name: us-east-1 - aws_secret_access_key: 'your-secret-key' - model: bedrock/us.anthropic.claude-sonnet-4-5-20250929-v1:0 - temperature: 1 - thinking: - budget_tokens: 10000 - type: enabled + aws_access_key_id: "your-access-key" + aws_region_name: us-east-1 + aws_secret_access_key: "your-secret-key" + model: bedrock/us.anthropic.claude-sonnet-4-5-20250929-v1:0 + temperature: 1 + thinking: + budget_tokens: 10000 + type: enabled azure-5: - api_base: https://your-resource.openai.azure.com - api_key: 'your-api-key' - api_version: 2025-01-01-preview - model: azure/gpt-5 - temperature: 0 + api_base: https://your-resource.openai.azure.com + api_key: "your-api-key" + api_version: 2025-01-01-preview + model: azure/gpt-5 + temperature: 0 ``` **2. Set the environment variable:** diff --git a/docs/installation/k9s-installation.md b/docs/installation/k9s-installation.md new file mode 100644 index 0000000000..f6f50be66a --- /dev/null +++ b/docs/installation/k9s-installation.md @@ -0,0 +1,138 @@ +# Install K9s + +## K9s Plugin + +Integrate HolmesGPT into your [K9s](https://github.com/derailed/k9s){:target="_blank"} Kubernetes terminal for instant analysis. + +![K9s Demo](../assets/K9sDemo.gif) + +### Prerequisites + +- **K9s must be installed** - See the [K9s installation guide](https://github.com/derailed/k9s#installation){:target="_blank"} +- **HolmesGPT CLI and API key** - Follow the [CLI Installation Guide](cli-installation.md) to install Holmes and configure your AI provider + +### Plugin Options + +??? note "Basic Plugin (Shift + H) - Quick investigation with predefined question" + + Add to your K9s plugins configuration file: + + - **Linux**: `~/.config/k9s/plugins.yaml` or `~/.k9s/plugins.yaml` + - **macOS**: `~/Library/Application Support/k9s/plugins.yaml` or `~/.k9s/plugins.yaml` + - **Windows**: `%APPDATA%/k9s/plugins.yaml` + + Read more about K9s plugins [here](https://k9scli.io/topics/plugins/){:target="_blank"} and check your plugin path [here](https://k9scli.io/topics/config/){:target="_blank"}. + + ```yaml + plugins: + holmesgpt: + shortCut: Shift-H + description: Ask HolmesGPT + scopes: + - all + command: bash + background: false + confirm: false + args: + - -c + - | + # Check if we're already using the correct context + CURRENT_CONTEXT=$(kubectl config current-context 2>/dev/null || echo "") + if [ "$CURRENT_CONTEXT" = "$CONTEXT" ]; then + # Already using the correct context, run HolmesGPT directly + holmes ask "why is $NAME of $RESOURCE_NAME in -n $NAMESPACE not working as expected" + else + # Create temporary kubeconfig to avoid changing user's system context + # K9s passes $CONTEXT but we need to ensure HolmesGPT uses the same context + # without permanently switching the user's kubectl context + TEMP_KUBECONFIG=$(mktemp) + kubectl config view --raw > $TEMP_KUBECONFIG + KUBECONFIG=$TEMP_KUBECONFIG kubectl config use-context $CONTEXT + # KUBECONFIG environment variable is passed to holmes and all its child processes + KUBECONFIG=$TEMP_KUBECONFIG holmes ask "why is $NAME of $RESOURCE_NAME in -n $NAMESPACE not working as expected" + rm -f $TEMP_KUBECONFIG + fi + echo "Press 'q' to exit" + while : ; do + read -n 1 k <&1 + if [[ $k = q ]] ; then + break + fi + done + ``` + +??? note "Advanced Plugin (Shift + Q) - Interactive plugin with custom questions" + + + Add to your K9s plugins configuration file: + + - **Linux**: `~/.config/k9s/plugins.yaml` or `~/.k9s/plugins.yaml` + - **macOS**: `~/Library/Application Support/k9s/plugins.yaml` or `~/.k9s/plugins.yaml` + - **Windows**: `%APPDATA%/k9s/plugins.yaml` + + Read more about K9s plugins [here](https://k9scli.io/topics/plugins/){:target="_blank"} and check your plugin path [here](https://k9scli.io/topics/config/){:target="_blank"}. + + ```yaml + plugins: + custom-holmesgpt: + shortCut: Shift-Q + description: Custom HolmesGPT Ask + scopes: + - all + command: bash + background: false + confirm: false + args: + - -c + - | + INSTRUCTIONS="# Edit the line below. Lines starting with '#' will be ignored." + DEFAULT_ASK_COMMAND="why is $NAME of $RESOURCE_NAME in -n $NAMESPACE not working as expected" + QUESTION_FILE=$(mktemp) + + echo "$INSTRUCTIONS" > "$QUESTION_FILE" + echo "$DEFAULT_ASK_COMMAND" >> "$QUESTION_FILE" + + # Open the line in the default text editor + ${EDITOR:-nano} "$QUESTION_FILE" + + # Read the modified line, ignoring lines starting with '#' + user_input=$(grep -v '^#' "$QUESTION_FILE") + + echo "Running: holmes ask '$user_input'" + # Check if we're already using the correct context + CURRENT_CONTEXT=$(kubectl config current-context 2>/dev/null || echo "") + if [ "$CURRENT_CONTEXT" = "$CONTEXT" ]; then + # Already using the correct context, run HolmesGPT directly + holmes ask "$user_input" + else + # Create temporary kubeconfig to avoid changing user's system context + # K9s passes $CONTEXT but we need to ensure HolmesGPT uses the same context + # without permanently switching the user's kubectl context + TEMP_KUBECONFIG=$(mktemp) + kubectl config view --raw > $TEMP_KUBECONFIG + KUBECONFIG=$TEMP_KUBECONFIG kubectl config use-context $CONTEXT + # KUBECONFIG environment variable is passed to holmes and all its child processes + KUBECONFIG=$TEMP_KUBECONFIG holmes ask "$user_input" + rm -f $TEMP_KUBECONFIG + fi + echo "Press 'q' to exit" + while : ; do + read -n 1 k <&1 + if [[ $k = q ]] ; then + break + fi + done + ``` + +### Usage + +1. Run K9s and select any Kubernetes resource +2. Press **Shift + H** for quick analysis or **Shift + Q** for custom questions + +## Need Help? + +- **[Join our Slack](https://cloud-native.slack.com/archives/C0A1SPQM5PZ){:target="_blank"}** - Get help from the community +- **[Request features on GitHub](https://github.com/HolmesGPT/holmesgpt/issues){:target="_blank"}** - Suggest improvements or report bugs +- **[Troubleshooting guide](../reference/troubleshooting.md)** - Common issues and solutions + + diff --git a/docs/installation/slack-installation.md b/docs/installation/slack-installation.md new file mode 100644 index 0000000000..c3e326101f --- /dev/null +++ b/docs/installation/slack-installation.md @@ -0,0 +1,19 @@ +# Install Slack Bot (third party) + +## Slack Bot (Robusta) + +First install Robusta SaaS, then tag HolmesGPT in any Slack message for instant analysis. + +![Robusta Slack Bot powered by Holmes](../assets/RobustaSlackBot-Poweredby-Holmes.png) + +### Setup Slack Bot + +[![Watch Slack Bot Demo](https://cdn.loom.com/sessions/thumbnails/7a60a42e854e45368e9b7f9d3c36ae5f-65bd123629db6922-full-play.gif)](https://www.loom.com/share/7a60a42e854e45368e9b7f9d3c36ae5f?sid=bfed9efb-b607-416c-b481-c2a63d314a4b){:target="_blank"} + +## Need Help? + +- **[Join our Slack](https://cloud-native.slack.com/archives/C0A1SPQM5PZ){:target="_blank"}** - Get help from the community +- **[Request features on GitHub](https://github.com/HolmesGPT/holmesgpt/issues){:target="_blank"}** - Suggest improvements or report bugs +- **[Troubleshooting guide](../reference/troubleshooting.md)** - Common issues and solutions + + diff --git a/docs/installation/ui-installation.md b/docs/installation/ui-installation.md index b9f4a0d85b..97b68bfe4e 100644 --- a/docs/installation/ui-installation.md +++ b/docs/installation/ui-installation.md @@ -1,136 +1,4 @@ -# Install UI/Slack/K9s - -Use HolmesGPT through graphical and terminal interfaces via third-party integrations. - -## K9s Plugin - -Integrate HolmesGPT into your [K9s](https://github.com/derailed/k9s){:target="_blank"} Kubernetes terminal for instant analysis. - -![K9s Demo](../assets/K9sDemo.gif) - -### Prerequisites - -- **K9s must be installed** - See the [K9s installation guide](https://github.com/derailed/k9s#installation){:target="_blank"} -- **HolmesGPT CLI and API key** - Follow the [CLI Installation Guide](cli-installation.md) to install Holmes and configure your AI provider - -### Plugin Options - -??? note "Basic Plugin (Shift + H) - Quick investigation with predefined question" - - Add to your K9s plugins configuration file: - - - **Linux**: `~/.config/k9s/plugins.yaml` or `~/.k9s/plugins.yaml` - - **macOS**: `~/Library/Application Support/k9s/plugins.yaml` or `~/.k9s/plugins.yaml` - - **Windows**: `%APPDATA%/k9s/plugins.yaml` - - Read more about K9s plugins [here](https://k9scli.io/topics/plugins/){:target="_blank"} and check your plugin path [here](https://k9scli.io/topics/config/){:target="_blank"}. - - ```yaml - plugins: - holmesgpt: - shortCut: Shift-H - description: Ask HolmesGPT - scopes: - - all - command: bash - background: false - confirm: false - args: - - -c - - | - # Check if we're already using the correct context - CURRENT_CONTEXT=$(kubectl config current-context 2>/dev/null || echo "") - if [ "$CURRENT_CONTEXT" = "$CONTEXT" ]; then - # Already using the correct context, run HolmesGPT directly - holmes ask "why is $NAME of $RESOURCE_NAME in -n $NAMESPACE not working as expected" - else - # Create temporary kubeconfig to avoid changing user's system context - # K9s passes $CONTEXT but we need to ensure HolmesGPT uses the same context - # without permanently switching the user's kubectl context - TEMP_KUBECONFIG=$(mktemp) - kubectl config view --raw > $TEMP_KUBECONFIG - KUBECONFIG=$TEMP_KUBECONFIG kubectl config use-context $CONTEXT - # KUBECONFIG environment variable is passed to holmes and all its child processes - KUBECONFIG=$TEMP_KUBECONFIG holmes ask "why is $NAME of $RESOURCE_NAME in -n $NAMESPACE not working as expected" - rm -f $TEMP_KUBECONFIG - fi - echo "Press 'q' to exit" - while : ; do - read -n 1 k <&1 - if [[ $k = q ]] ; then - break - fi - done - ``` - -??? note "Advanced Plugin (Shift + Q) - Interactive plugin with custom questions" - - - Add to your K9s plugins configuration file: - - - **Linux**: `~/.config/k9s/plugins.yaml` or `~/.k9s/plugins.yaml` - - **macOS**: `~/Library/Application Support/k9s/plugins.yaml` or `~/.k9s/plugins.yaml` - - **Windows**: `%APPDATA%/k9s/plugins.yaml` - - Read more about K9s plugins [here](https://k9scli.io/topics/plugins/){:target="_blank"} and check your plugin path [here](https://k9scli.io/topics/config/){:target="_blank"}. - - ```yaml - plugins: - custom-holmesgpt: - shortCut: Shift-Q - description: Custom HolmesGPT Ask - scopes: - - all - command: bash - background: false - confirm: false - args: - - -c - - | - INSTRUCTIONS="# Edit the line below. Lines starting with '#' will be ignored." - DEFAULT_ASK_COMMAND="why is $NAME of $RESOURCE_NAME in -n $NAMESPACE not working as expected" - QUESTION_FILE=$(mktemp) - - echo "$INSTRUCTIONS" > "$QUESTION_FILE" - echo "$DEFAULT_ASK_COMMAND" >> "$QUESTION_FILE" - - # Open the line in the default text editor - ${EDITOR:-nano} "$QUESTION_FILE" - - # Read the modified line, ignoring lines starting with '#' - user_input=$(grep -v '^#' "$QUESTION_FILE") - - echo "Running: holmes ask '$user_input'" - # Check if we're already using the correct context - CURRENT_CONTEXT=$(kubectl config current-context 2>/dev/null || echo "") - if [ "$CURRENT_CONTEXT" = "$CONTEXT" ]; then - # Already using the correct context, run HolmesGPT directly - holmes ask "$user_input" - else - # Create temporary kubeconfig to avoid changing user's system context - # K9s passes $CONTEXT but we need to ensure HolmesGPT uses the same context - # without permanently switching the user's kubectl context - TEMP_KUBECONFIG=$(mktemp) - kubectl config view --raw > $TEMP_KUBECONFIG - KUBECONFIG=$TEMP_KUBECONFIG kubectl config use-context $CONTEXT - # KUBECONFIG environment variable is passed to holmes and all its child processes - KUBECONFIG=$TEMP_KUBECONFIG holmes ask "$user_input" - rm -f $TEMP_KUBECONFIG - fi - echo "Press 'q' to exit" - while : ; do - read -n 1 k <&1 - if [[ $k = q ]] ; then - break - fi - done - ``` - -### Usage - -1. Run K9s and select any Kubernetes resource -2. Press **Shift + H** for quick analysis or **Shift + Q** for custom questions - +# Install UI (third party) ## Web UI (Robusta) @@ -173,20 +41,10 @@ The fastest way to use HolmesGPT is via the managed Robusta SaaS platform. !!! tip "Multiple AI Providers" You can configure multiple AI models for users to choose from in the UI. See [Using Multiple Providers](../ai-providers/using-multiple-providers.md) for configuration details. ---- - -## Slack Bot (Robusta) - -First install Robusta SaaS, then tag HolmesGPT in any Slack message for instant analysis. - -![Robusta Slack Bot powered by Holmes](../assets/RobustaSlackBot-Poweredby-Holmes.png) - -### Setup Slack Bot - -[![Watch Slack Bot Demo](https://cdn.loom.com/sessions/thumbnails/7a60a42e854e45368e9b7f9d3c36ae5f-65bd123629db6922-full-play.gif)](https://www.loom.com/share/7a60a42e854e45368e9b7f9d3c36ae5f?sid=bfed9efb-b607-416c-b481-c2a63d314a4b){:target="_blank"} - ## Need Help? - **[Join our Slack](https://cloud-native.slack.com/archives/C0A1SPQM5PZ){:target="_blank"}** - Get help from the community - **[Request features on GitHub](https://github.com/HolmesGPT/holmesgpt/issues){:target="_blank"}** - Suggest improvements or report bugs - **[Troubleshooting guide](../reference/troubleshooting.md)** - Common issues and solutions + + diff --git a/mkdocs.yml b/mkdocs.yml index 7978d610fc..416c986c08 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -10,6 +10,68 @@ exclude_docs: | README.md cncf-self-assesment.md +nav: + - Home: index.md + - Installation: + - 'Install CLI': installation/cli-installation.md + - 'Install UI (third party)': installation/ui-installation.md + - 'Install Slack Bot (third party)': installation/slack-installation.md + - 'Install K9s': installation/k9s-installation.md + - 'Install Backstage (third party)': installation/backstage-installation.md + - 'Install Helm Chart': installation/kubernetes-installation.md + - 'Install Python SDK': installation/python-installation.md + - Walkthrough: + - 'Overview': walkthrough/index.md + - 'Interactive Mode': walkthrough/interactive-mode.md + - 'CI/CD Troubleshooting': walkthrough/cicd-troubleshooting.md + - 'Investigating Prometheus Alerts': walkthrough/investigating-prometheus-alerts.md + - 'Investigating using AKS MCP Server': walkthrough/investigating-using-aks-mcp-server.md + - 'AI Providers': + - 'Overview': ai-providers/index.md + - Anthropic: ai-providers/anthropic.md + - 'AWS Bedrock': ai-providers/aws-bedrock.md + - 'Azure OpenAI': ai-providers/azure-openai.md + - Gemini: ai-providers/gemini.md + - 'Google Vertex AI': ai-providers/google-vertex-ai.md + - Ollama: ai-providers/ollama.md + - OpenAI: ai-providers/openai.md + - 'OpenAI-Compatible': ai-providers/openai-compatible.md + - Other: ai-providers/other.md + - 'Robusta AI': ai-providers/robusta-ai.md + - 'Using Multiple Providers': ai-providers/using-multiple-providers.md + - 'Data Sources': + - 'Overview': data-sources/index.md + - 'Built-in Toolsets': data-sources/builtin-toolsets/index.md + - 'Custom Toolsets': data-sources/custom-toolsets.md + - 'MCP Servers': data-sources/remote-mcp-servers.md + - 'Adding Permissions for Additional Resources': data-sources/permissions.md + - Benchmarks: + - 'Latest Results': development/evaluations/latest-results.md + - 'Historical Results': + - 'Overview': development/evaluations/history/index.md + - 'September 28, 2025': development/evaluations/history/results_20250928_001434.md + - 'September 30, 2025 (custom Claude)': development/evaluations/history/custom_claude_results_20250930_153753.md + - 'September 30, 2025': development/evaluations/history/results_20250930_085923.md + - 'October 12, 2025': development/evaluations/history/results_20251012_170303.md + - 'Self-Hosted Models v1': development/evaluations/history/custom_self_hosted_results_20251008_053744.md + - 'November 27, 2025': development/evaluations/history/results_20251127_042958.md + - 'Running Evaluations': development/evaluations/running-evals.md + - 'Adding New Evaluations': development/evaluations/adding-evals.md + - 'Benchmarking New Models': development/evaluations/benchmarking-new-models.md + - 'Reporting with Braintrust': development/evaluations/reporting.md + - Development: + - 'Overview': development/index.md + - 'Tool Output Transformers': development/transformers.md + - Reference: + - 'Environment Variables': reference/environment-variables.md + - 'Helm Configuration': reference/helm-configuration.md + - 'Kubernetes Permissions': reference/kubernetes-permissions.md + - 'HTTP API': reference/http-api.md + - 'Slash Commands': reference/slash-commands.md + - 'Troubleshooting': reference/troubleshooting.md + - Community: community.md + - 'CNCF Self Assessment': cncf-self-assesment.md + theme: name: material custom_dir: docs/overrides From cea4fff32e687e490d257adf83279ce5296d21a2 Mon Sep 17 00:00:00 2001 From: Filip Grebowski Date: Sat, 27 Dec 2025 13:13:48 +0000 Subject: [PATCH 05/27] mkdocs page order Signed-off-by: Filip Grebowski --- mkdocs.yml | 366 ++++++++++++++++++++++++++--------------------------- 1 file changed, 183 insertions(+), 183 deletions(-) diff --git a/mkdocs.yml b/mkdocs.yml index 416c986c08..26e3a0bc24 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -5,200 +5,200 @@ repo_url: https://github.com/HolmesGPT/holmesgpt repo_name: HolmesGPT/holmesgpt edit_uri: edit/master/docs/ exclude_docs: | - _* - snippets/ - README.md - cncf-self-assesment.md + _* + snippets/ + README.md + cncf-self-assesment.md nav: - - Home: index.md - - Installation: - - 'Install CLI': installation/cli-installation.md - - 'Install UI (third party)': installation/ui-installation.md - - 'Install Slack Bot (third party)': installation/slack-installation.md - - 'Install K9s': installation/k9s-installation.md - - 'Install Backstage (third party)': installation/backstage-installation.md - - 'Install Helm Chart': installation/kubernetes-installation.md - - 'Install Python SDK': installation/python-installation.md - - Walkthrough: - - 'Overview': walkthrough/index.md - - 'Interactive Mode': walkthrough/interactive-mode.md - - 'CI/CD Troubleshooting': walkthrough/cicd-troubleshooting.md - - 'Investigating Prometheus Alerts': walkthrough/investigating-prometheus-alerts.md - - 'Investigating using AKS MCP Server': walkthrough/investigating-using-aks-mcp-server.md - - 'AI Providers': - - 'Overview': ai-providers/index.md - - Anthropic: ai-providers/anthropic.md - - 'AWS Bedrock': ai-providers/aws-bedrock.md - - 'Azure OpenAI': ai-providers/azure-openai.md - - Gemini: ai-providers/gemini.md - - 'Google Vertex AI': ai-providers/google-vertex-ai.md - - Ollama: ai-providers/ollama.md - - OpenAI: ai-providers/openai.md - - 'OpenAI-Compatible': ai-providers/openai-compatible.md - - Other: ai-providers/other.md - - 'Robusta AI': ai-providers/robusta-ai.md - - 'Using Multiple Providers': ai-providers/using-multiple-providers.md - - 'Data Sources': - - 'Overview': data-sources/index.md - - 'Built-in Toolsets': data-sources/builtin-toolsets/index.md - - 'Custom Toolsets': data-sources/custom-toolsets.md - - 'MCP Servers': data-sources/remote-mcp-servers.md - - 'Adding Permissions for Additional Resources': data-sources/permissions.md - - Benchmarks: - - 'Latest Results': development/evaluations/latest-results.md - - 'Historical Results': - - 'Overview': development/evaluations/history/index.md - - 'September 28, 2025': development/evaluations/history/results_20250928_001434.md - - 'September 30, 2025 (custom Claude)': development/evaluations/history/custom_claude_results_20250930_153753.md - - 'September 30, 2025': development/evaluations/history/results_20250930_085923.md - - 'October 12, 2025': development/evaluations/history/results_20251012_170303.md - - 'Self-Hosted Models v1': development/evaluations/history/custom_self_hosted_results_20251008_053744.md - - 'November 27, 2025': development/evaluations/history/results_20251127_042958.md - - 'Running Evaluations': development/evaluations/running-evals.md - - 'Adding New Evaluations': development/evaluations/adding-evals.md - - 'Benchmarking New Models': development/evaluations/benchmarking-new-models.md - - 'Reporting with Braintrust': development/evaluations/reporting.md - - Development: - - 'Overview': development/index.md - - 'Tool Output Transformers': development/transformers.md - - Reference: - - 'Environment Variables': reference/environment-variables.md - - 'Helm Configuration': reference/helm-configuration.md - - 'Kubernetes Permissions': reference/kubernetes-permissions.md - - 'HTTP API': reference/http-api.md - - 'Slash Commands': reference/slash-commands.md - - 'Troubleshooting': reference/troubleshooting.md - - Community: community.md - - 'CNCF Self Assessment': cncf-self-assesment.md + - Home: index.md + - Installation: + - "Install CLI": installation/cli-installation.md + - "Install UI (third party)": installation/ui-installation.md + - "Install Slack Bot (third party)": installation/slack-installation.md + - "Install Backstage (third party)": installation/backstage-installation.md + - "Install K9s": installation/k9s-installation.md + - "Install Helm Chart": installation/kubernetes-installation.md + - "Install Python SDK": installation/python-installation.md + - Walkthrough: + - "Overview": walkthrough/index.md + - "Interactive Mode": walkthrough/interactive-mode.md + - "CI/CD Troubleshooting": walkthrough/cicd-troubleshooting.md + - "Investigating Prometheus Alerts": walkthrough/investigating-prometheus-alerts.md + - "Investigating using AKS MCP Server": walkthrough/investigating-using-aks-mcp-server.md + - "AI Providers": + - "Overview": ai-providers/index.md + - Anthropic: ai-providers/anthropic.md + - "AWS Bedrock": ai-providers/aws-bedrock.md + - "Azure OpenAI": ai-providers/azure-openai.md + - Gemini: ai-providers/gemini.md + - "Google Vertex AI": ai-providers/google-vertex-ai.md + - Ollama: ai-providers/ollama.md + - OpenAI: ai-providers/openai.md + - "OpenAI-Compatible": ai-providers/openai-compatible.md + - Other: ai-providers/other.md + - "Robusta AI": ai-providers/robusta-ai.md + - "Using Multiple Providers": ai-providers/using-multiple-providers.md + - "Data Sources": + - "Overview": data-sources/index.md + - "Built-in Toolsets": data-sources/builtin-toolsets/index.md + - "Custom Toolsets": data-sources/custom-toolsets.md + - "MCP Servers": data-sources/remote-mcp-servers.md + - "Adding Permissions for Additional Resources": data-sources/permissions.md + - Benchmarks: + - "Latest Results": development/evaluations/latest-results.md + - "Historical Results": + - "Overview": development/evaluations/history/index.md + - "September 28, 2025": development/evaluations/history/results_20250928_001434.md + - "September 30, 2025 (custom Claude)": development/evaluations/history/custom_claude_results_20250930_153753.md + - "September 30, 2025": development/evaluations/history/results_20250930_085923.md + - "October 12, 2025": development/evaluations/history/results_20251012_170303.md + - "Self-Hosted Models v1": development/evaluations/history/custom_self_hosted_results_20251008_053744.md + - "November 27, 2025": development/evaluations/history/results_20251127_042958.md + - "Running Evaluations": development/evaluations/running-evals.md + - "Adding New Evaluations": development/evaluations/adding-evals.md + - "Benchmarking New Models": development/evaluations/benchmarking-new-models.md + - "Reporting with Braintrust": development/evaluations/reporting.md + - Development: + - "Overview": development/index.md + - "Tool Output Transformers": development/transformers.md + - Reference: + - "Environment Variables": reference/environment-variables.md + - "Helm Configuration": reference/helm-configuration.md + - "Kubernetes Permissions": reference/kubernetes-permissions.md + - "HTTP API": reference/http-api.md + - "Slash Commands": reference/slash-commands.md + - "Troubleshooting": reference/troubleshooting.md + - Community: community.md + - "CNCF Self Assessment": cncf-self-assesment.md theme: - name: material - custom_dir: docs/overrides - features: - - announce.dismiss - - content.action.edit - - content.action.view - - content.code.annotate - - content.code.copy - - content.tooltips - - navigation.footer - - navigation.indexes - - navigation.instant - - navigation.instant.prefetch - - navigation.path - - navigation.top - - navigation.tracking - - search.highlight - - search.share - - search.suggest - - toc.follow - palette: - - media: "(prefers-color-scheme)" - toggle: - icon: material/brightness-auto - name: Switch to light mode - - media: "(prefers-color-scheme: light)" - scheme: default - primary: teal - accent: teal - toggle: - icon: material/brightness-7 - name: Switch to dark mode - - media: "(prefers-color-scheme: dark)" - scheme: slate - primary: teal - accent: teal - toggle: - icon: material/brightness-4 - name: Switch to system preference - font: - text: Inter - code: Fira Code - favicon: assets/favicon.png - logo: assets/logo.png + name: material + custom_dir: docs/overrides + features: + - announce.dismiss + - content.action.edit + - content.action.view + - content.code.annotate + - content.code.copy + - content.tooltips + - navigation.footer + - navigation.indexes + - navigation.instant + - navigation.instant.prefetch + - navigation.path + - navigation.top + - navigation.tracking + - search.highlight + - search.share + - search.suggest + - toc.follow + palette: + - media: "(prefers-color-scheme)" + toggle: + icon: material/brightness-auto + name: Switch to light mode + - media: "(prefers-color-scheme: light)" + scheme: default + primary: teal + accent: teal + toggle: + icon: material/brightness-7 + name: Switch to dark mode + - media: "(prefers-color-scheme: dark)" + scheme: slate + primary: teal + accent: teal + toggle: + icon: material/brightness-4 + name: Switch to system preference + font: + text: Inter + code: Fira Code + favicon: assets/favicon.png + logo: assets/logo.png copyright: | - Copyright HolmesGPT a Series of LF Projects, LLC - For website terms of use, trademark policy and other project policies please see lfprojects.org/policies/. + Copyright HolmesGPT a Series of LF Projects, LLC + For website terms of use, trademark policy and other project policies please see lfprojects.org/policies/. plugins: - - awesome-nav - - search: - separator: '[\s\-,:!=\[\]()"`/]+|\.(?!\d)|&[lg]t;|(?!\b)(?=[A-Z][a-z])' - - glightbox - # - git-revision-date-localized: - # enable_creation_date: true - # type: timeago - # - git-committers: - # repository: HolmesGPT/holmesgpt - # branch: master - # docs_path: docs/ - # - minify: - # minify_html: true + - awesome-nav + - search: + separator: '[\s\-,:!=\[\]()"`/]+|\.(?!\d)|&[lg]t;|(?!\b)(?=[A-Z][a-z])' + - glightbox + # - git-revision-date-localized: + # enable_creation_date: true + # type: timeago + # - git-committers: + # repository: HolmesGPT/holmesgpt + # branch: master + # docs_path: docs/ + # - minify: + # minify_html: true markdown_extensions: - - abbr - - admonition - - attr_list - - def_list - - footnotes - - md_in_html - - toc: - permalink: true - - pymdownx.arithmatex: - generic: true - - pymdownx.betterem: - smart_enable: all - - pymdownx.caret - - pymdownx.details - - pymdownx.emoji: - emoji_generator: !!python/name:material.extensions.emoji.to_svg - emoji_index: !!python/name:material.extensions.emoji.twemoji - - pymdownx.highlight: - anchor_linenums: true - line_spans: __span - pygments_lang_class: true - - pymdownx.inlinehilite - - pymdownx.keys - - pymdownx.magiclink: - normalize_issue_symbols: true - repo_url_shorthand: true - user: robusta-dev - repo: holmesgpt - - pymdownx.mark - - pymdownx.smartsymbols - - pymdownx.snippets: - base_path: docs - - pymdownx.superfences: - custom_fences: - - name: mermaid - class: mermaid - format: !!python/name:pymdownx.superfences.fence_code_format - - name: yaml-helm-values - class: yaml - format: !!python/name:docs.custom_fences.helm_tabs_fence_format - - name: yaml-toolset-config - class: yaml - format: !!python/name:docs.custom_fences.toolset_config_fence_format - - pymdownx.tabbed: - alternate_style: true - combine_header_slug: true - slugify: !!python/object/apply:pymdownx.slugs.slugify - kwds: - case: lower - - pymdownx.tasklist: - custom_checkbox: true - - pymdownx.tilde + - abbr + - admonition + - attr_list + - def_list + - footnotes + - md_in_html + - toc: + permalink: true + - pymdownx.arithmatex: + generic: true + - pymdownx.betterem: + smart_enable: all + - pymdownx.caret + - pymdownx.details + - pymdownx.emoji: + emoji_generator: !!python/name:material.extensions.emoji.to_svg + emoji_index: !!python/name:material.extensions.emoji.twemoji + - pymdownx.highlight: + anchor_linenums: true + line_spans: __span + pygments_lang_class: true + - pymdownx.inlinehilite + - pymdownx.keys + - pymdownx.magiclink: + normalize_issue_symbols: true + repo_url_shorthand: true + user: robusta-dev + repo: holmesgpt + - pymdownx.mark + - pymdownx.smartsymbols + - pymdownx.snippets: + base_path: docs + - pymdownx.superfences: + custom_fences: + - name: mermaid + class: mermaid + format: !!python/name:pymdownx.superfences.fence_code_format + - name: yaml-helm-values + class: yaml + format: !!python/name:docs.custom_fences.helm_tabs_fence_format + - name: yaml-toolset-config + class: yaml + format: !!python/name:docs.custom_fences.toolset_config_fence_format + - pymdownx.tabbed: + alternate_style: true + combine_header_slug: true + slugify: !!python/object/apply:pymdownx.slugs.slugify + kwds: + case: lower + - pymdownx.tasklist: + custom_checkbox: true + - pymdownx.tilde extra_css: - - stylesheets/extra.css + - stylesheets/extra.css extra: - social: - - icon: fontawesome/brands/github - link: https://github.com/HolmesGPT/holmesgpt - - icon: fontawesome/brands/slack - link: https://cloud-native.slack.com/archives/C0A1SPQM5PZ - - icon: fontawesome/brands/twitter - link: https://twitter.com/RobustaDev + social: + - icon: fontawesome/brands/github + link: https://github.com/HolmesGPT/holmesgpt + - icon: fontawesome/brands/slack + link: https://cloud-native.slack.com/archives/C0A1SPQM5PZ + - icon: fontawesome/brands/twitter + link: https://twitter.com/RobustaDev From 5a7e9f611f5b140426c89a617f7501d612ed2e8c Mon Sep 17 00:00:00 2001 From: moshemorad Date: Sun, 21 Dec 2025 15:17:17 +0200 Subject: [PATCH 06/27] Add runbook alerts to prompt (#1205) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 1. Add runbook alerts to runbook prompts so llm will be able to select it. 2. Remove old alert runbook fetching as part of invesigation 3. Add eval for alert with multiple relevant runbooks due to matching alert - Test report here: https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/root-master-k%3D165_-20251218_145120?c= 4. Removed responses logs that spammed test logs. ## Summary by CodeRabbit # Release Notes * **New Features** * Enhanced runbook matching to include alert-based matching in addition to symptom-based matching, improving runbook discovery accuracy. * Added alerts metadata to runbook catalog entries for better runbook relevance filtering. * **Chores** * Simplified investigation workflow by streamlining instruction data flow. * Improved test coverage with new integration test for multi-runbook alert scenarios. ✏️ Tip: You can customize this high-level summary in your review settings. --------- Signed-off-by: Mohse Morad Signed-off-by: Filip Grebowski --- holmes/core/investigation.py | 9 -- holmes/core/tool_calling_llm.py | 3 - holmes/main.py | 5 - .../prompts/_runbook_instructions.jinja2 | 3 +- holmes/plugins/runbooks/__init__.py | 5 +- tests/conftest.py | 7 ++ .../issue_data.json | 93 +++++++++++++++++++ .../pod_not_ready.yaml | 18 ++++ .../runbook_catalog.json | 23 +++++ ..._40296b5c-2cb5-41df-b1f5-93441f894c44.json | 6 ++ ..._8fe8e24d-6b53-47a6-92a5-b389938fa823.json | 9 ++ ..._b2ffd311-f339-46a2-b379-1e53eb0bc1ed.json | 9 ++ .../test_case.yaml | 16 ++++ tests/llm/utils/mock_dal.py | 2 +- tests/llm/utils/test_case_utils.py | 2 +- tests/test_issue_investigator.py | 8 -- 16 files changed, 188 insertions(+), 30 deletions(-) create mode 100644 tests/llm/fixtures/test_ask_holmes/165_alert_with_multiple_runbooks/issue_data.json create mode 100644 tests/llm/fixtures/test_ask_holmes/165_alert_with_multiple_runbooks/pod_not_ready.yaml create mode 100644 tests/llm/fixtures/test_ask_holmes/165_alert_with_multiple_runbooks/runbook_catalog.json create mode 100644 tests/llm/fixtures/test_ask_holmes/165_alert_with_multiple_runbooks/runbook_content_40296b5c-2cb5-41df-b1f5-93441f894c44.json create mode 100644 tests/llm/fixtures/test_ask_holmes/165_alert_with_multiple_runbooks/runbook_content_8fe8e24d-6b53-47a6-92a5-b389938fa823.json create mode 100644 tests/llm/fixtures/test_ask_holmes/165_alert_with_multiple_runbooks/runbook_content_b2ffd311-f339-46a2-b379-1e53eb0bc1ed.json create mode 100644 tests/llm/fixtures/test_ask_holmes/165_alert_with_multiple_runbooks/test_case.yaml diff --git a/holmes/core/investigation.py b/holmes/core/investigation.py index 4ee187b00e..a1cdaa3d35 100644 --- a/holmes/core/investigation.py +++ b/holmes/core/investigation.py @@ -33,9 +33,6 @@ def investigate_issues( ) -> InvestigationResult: context = dal.get_issue_data(investigate_request.context.get("robusta_issue_id")) - resource_instructions = dal.get_resource_instructions( - "alert", investigate_request.context.get("issue_type") - ) global_instructions = dal.get_global_instructions_for_account() raw_data = investigate_request.model_dump() @@ -61,7 +58,6 @@ def investigate_issues( issue, prompt=investigate_request.prompt_template, post_processing_prompt=HOLMES_POST_PROCESSING_PROMPT, - instructions=resource_instructions, global_instructions=global_instructions, sections=investigate_request.sections, trace_span=trace_span, @@ -106,10 +102,6 @@ def get_investigation_context( issue_instructions = ai.runbook_manager.get_instructions_for_issue(issue) - resource_instructions = dal.get_resource_instructions( - "alert", investigate_request.context.get("issue_type") - ) - # This section is about setting vars to request the LLM to return structured output. # It does not mean that Holmes will not return structured sections for investigation as it is # capable of splitting the markdown into sections @@ -152,7 +144,6 @@ def get_investigation_context( runbook_catalog=runbook_catalog, global_instructions=global_instructions, issue_instructions=issue_instructions, - resource_instructions=resource_instructions, ) base_user = f"{base_user}\n #This is context from the issue:\n{issue.raw}" diff --git a/holmes/core/tool_calling_llm.py b/holmes/core/tool_calling_llm.py index 95a53be022..8b04847ccd 100644 --- a/holmes/core/tool_calling_llm.py +++ b/holmes/core/tool_calling_llm.py @@ -34,7 +34,6 @@ ) from holmes.core.issue import Issue from holmes.core.llm import LLM -from holmes.core.resource_instruction import ResourceInstructions from holmes.core.runbooks import RunbookManager from holmes.core.safeguards import prevent_overly_repeated_tool_call from holmes.core.tools import ( @@ -1076,7 +1075,6 @@ def investigate( self, issue: Issue, prompt: str, - instructions: Optional[ResourceInstructions], console: Optional[Console] = None, global_instructions: Optional[Instructions] = None, post_processing_prompt: Optional[str] = None, @@ -1140,7 +1138,6 @@ def investigate( runbook_catalog=runbooks, global_instructions=global_instructions, issue_instructions=issue_runbooks, - resource_instructions=instructions, ) user_prompt = generate_user_prompt( base_user, diff --git a/holmes/main.py b/holmes/main.py index 5baf234103..04c91b0fbb 100644 --- a/holmes/main.py +++ b/holmes/main.py @@ -449,7 +449,6 @@ def alertmanager( issue=issue, prompt=system_prompt, # type: ignore console=console, - instructions=None, post_processing_prompt=post_processing_prompt, ) results.append({"issue": issue.model_dump(), "result": result.model_dump()}) @@ -566,7 +565,6 @@ def jira( issue=issue, prompt=system_prompt, # type: ignore console=console, - instructions=None, post_processing_prompt=post_processing_prompt, ) @@ -758,7 +756,6 @@ def github( issue=issue, prompt=system_prompt, # type: ignore console=console, - instructions=None, post_processing_prompt=post_processing_prompt, ) @@ -844,7 +841,6 @@ def pagerduty( issue=issue, prompt=system_prompt, # type: ignore console=console, - instructions=None, post_processing_prompt=post_processing_prompt, ) @@ -927,7 +923,6 @@ def opsgenie( issue=issue, prompt=system_prompt, # type: ignore console=console, - instructions=None, post_processing_prompt=post_processing_prompt, ) diff --git a/holmes/plugins/prompts/_runbook_instructions.jinja2 b/holmes/plugins/prompts/_runbook_instructions.jinja2 index be16ffa23c..1b5e4f03e2 100644 --- a/holmes/plugins/prompts/_runbook_instructions.jinja2 +++ b/holmes/plugins/prompts/_runbook_instructions.jinja2 @@ -7,7 +7,8 @@ {%- if available -%} # Runbook Selection -You (HolmesGPT) have access to runbooks with step-by-step troubleshooting instructions. If one of the following runbooks relates to the user's issue, you MUST fetch it with the fetch_runbook tool. +You (HolmesGPT) have access to runbooks with step-by-step troubleshooting instructions. +If one of the following runbooks relates to the user's issue or match one of the alerts or symptoms listed in the runbook entry, you MUST fetch it with the fetch_runbook tool. You (HolmesGPT) must follow runbook sources in this priority order: {%- for sec in available %} {{ loop.index }}) {{ sec.title }} (priority #{{ loop.index }}) diff --git a/holmes/plugins/runbooks/__init__.py b/holmes/plugins/runbooks/__init__.py index bdd9e9c632..e3fe82d338 100644 --- a/holmes/plugins/runbooks/__init__.py +++ b/holmes/plugins/runbooks/__init__.py @@ -24,6 +24,7 @@ class RobustaRunbookInstruction(BaseModel): symptom: str title: str instruction: Optional[str] = None + alerts: List[str] = [] """ Custom YAML dumper to represent multi-line strings in literal block style due to instructions often being multi-line. @@ -54,7 +55,7 @@ def to_list_string(self) -> str: return f"{self.id}" def to_prompt_string(self) -> str: - return f"id='{self.id}' | title='{self.title}' | symptom='{self.symptom}'" + return f"id='{self.id}' | title='{self.title}' | symptom='{self.symptom}' | relevant alerts={', '.join(self.alerts)}" def pretty(self) -> str: try: @@ -149,7 +150,7 @@ def to_prompt_string(self) -> str: parts.append("Here are MD runbooks:") parts.extend(f"* {e.to_prompt_string()}" for e in md) if robusta: - parts.append("Here are Robusta runbooks:") + parts.append("\nHere are Robusta runbooks:") parts.extend(f"* {e.to_prompt_string()}" for e in robusta) return "\n".join(parts) diff --git a/tests/conftest.py b/tests/conftest.py index b48993ae70..0a2b7225c3 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -2,6 +2,7 @@ from typing import Any, Optional import pytest import yaml +import logging from holmes.config import Config from holmes.core.llm import LLM, TokenCountMetadata @@ -35,6 +36,12 @@ } +@pytest.fixture(autouse=True, scope="session") +def setup_logging(): + """Setup logging for the test session.""" + logging.getLogger("responses").setLevel(logging.WARNING) + + @pytest.fixture(autouse=True, scope="function") def clear_all_caches(): """Clear all function caches that may affect test isolation.""" diff --git a/tests/llm/fixtures/test_ask_holmes/165_alert_with_multiple_runbooks/issue_data.json b/tests/llm/fixtures/test_ask_holmes/165_alert_with_multiple_runbooks/issue_data.json new file mode 100644 index 0000000000..6d067ab09d --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/165_alert_with_multiple_runbooks/issue_data.json @@ -0,0 +1,93 @@ +{ + "id": "8604ba7a-2e54-4d1d-ae92-e1d1cfea7b98", + "description": "Readiness probe failed", + "source": "prometheus", + "category": null, + "priority": "INFO", + "account_id": "caa68d87-fa5b-4f0e-a6ca-a6514ccb11eb", + "subject_type": "pod", + "subject_name": "search-engine-service", + "service_key": "", + "subject_namespace": "default", + "cluster": "eu-prod-atc-aks", + "creation_date": "2025-08-23T12:47:39.0171", + "title": "Readiness probe failed:", + "aggregation_key": "ProbeFailure", + "finding_type": "issue", + "failure": true, + "fingerprint": "2e2665efb18e8453", + "group_id": null, + "subject_node": "aks-agentpool-60973701-vmss000001", + "starts_at": "2025-08-19T08:46:46.16+00:00", + "updated_at": "2025-08-23T12:47:38.924777+00:00", + "ends_at": null, + "service_kind": null, + "labels": { + "pod": "search-engine-service", + "service": "robusta-kube-prometheus-st-kubelet", + "instance": "10.224.0.5:10250", + "severity": "info", + "agentpool": "agentpool", + "alertname": "ProbeFailure", + "container": "stress", + "namespace": "default", + "prometheus": "monitoring/robusta-kube-prometheus-st-prometheus", + "kubernetes.io/os": "linux", + "kubernetes.io/arch": "amd64", + "beta.kubernetes.io/os": "linux", + "kubernetes.io/hostname": "aks-agentpool-60973701-vmss000001", + "beta.kubernetes.io/arch": "amd64", + "kubernetes.azure.com/mode": "system", + "kubernetes.azure.com/role": "agent", + "kubernetes.azure.com/os-sku": "Ubuntu", + "topology.kubernetes.io/zone": "0", + "kubernetes.azure.com/cluster": "MC_avi-resources_avi-test-cluster2_swedencentral", + "topology.kubernetes.io/region": "swedencentral", + "kubernetes.azure.com/agentpool": "agentpool", + "beta.kubernetes.io/instance-type": "Standard_D4ds_v5", + "node.kubernetes.io/instance-type": "Standard_D4ds_v5", + "topology.disk.csi.azure.com/zone": "", + "kubernetes.azure.com/network-name": "aks-vnet-38960791", + "kubernetes.azure.com/nodepool-type": "VirtualMachineScaleSets", + "kubernetes.azure.com/network-policy": "none", + "kubernetes.azure.com/network-subnet": "aks-subnet", + "kubernetes.azure.com/podnetwork-type": "overlay", + "kubernetes.azure.com/os-sku-effective": "Ubuntu2204", + "kubernetes.azure.com/os-sku-requested": "Ubuntu", + "failure-domain.beta.kubernetes.io/zone": "0", + "kubernetes.azure.com/azure-cni-overlay": "true", + "kubernetes.azure.com/kubelet-serving-ca": "cluster", + "kubernetes.azure.com/node-image-version": "AKSUbuntu-2204gen2containerd-202507.21.0", + "failure-domain.beta.kubernetes.io/region": "swedencentral", + "kubernetes.azure.com/network-subscription": "e7a7e3c5-ff48-4ccb-898b-83aa5d2f9097", + "kubernetes.azure.com/nodenetwork-vnetguid": "67eac630-110b-486f-b0ff-270fca469c6b", + "kubernetes.azure.com/network-resourcegroup": "avi-resources", + "kubernetes.azure.com/kubelet-identity-client-id": "ef0dac03-4262-4ed9-a591-4309704795e4", + "kubernetes.azure.com/consolidated-additional-properties": "8de655ff-6ffd-11f0-9af4-06b3f76bdbd5" + }, + "annotations": { + "summary": "Probe failure", + "node.alpha.kubernetes.io/ttl": "0", + "csi.volume.kubernetes.io/nodeid": "{\"disk.csi.azure.com\":\"aks-agentpool-60973701-vmss000001\",\"file.csi.azure.com\":\"aks-agentpool-60973701-vmss000001\"}", + "alpha.kubernetes.io/provided-node-ip": "10.224.0.5", + "volumes.kubernetes.io/controller-managed-attach-detach": "true" + }, + "evidence": [ + { + "file_type": "structured_data", + "data": "[{\"type\": \"markdown\", \"data\": \"**Alert labels**\"}, {\"type\": \"table\", \"data\": {\"headers\": [\"label\", \"value\"], \"rows\": [[\"alertname\", \"ProbeFailure\"], [\"container\", \"stress\"], [\"instance\", \"10.224.0.5:10250\"], [\"namespace\", \"default\"], [\"pod\", \"search-engine-service\"], [\"prometheus\", \"monitoring/robusta-kube-prometheus-st-prometheus\"], [\"service\", \"robusta-kube-prometheus-st-kubelet\"], [\"severity\", \"info\"]], \"column_renderers\": {}}, \"metadata\": {\"format\": \"vertical\"}}]", + "creation_date": "2025-08-23T12:47:38.846075", + "issue_id": "8604ba7a-2e54-4d1d-ae92-e1d1cfea7b98", + "id": "f4969841-602b-4680-bc54-9c9c79e1675a", + "account_id": "caa68d87-fa5b-4f0e-a6ca-a6514ccb11eb", + "enrichment_type": "alert_labels", + "collection_timestamp": null, + "title": "Alert labels" + } + ], + "start_timestamp": "2025-08-19T08:36:46.160000Z", + "end_timestamp": "2025-08-19T08:56:46.160000Z", + "start_timestamp_millis": 1755592606160, + "end_timestamp_millis": 1755593806160 + } + diff --git a/tests/llm/fixtures/test_ask_holmes/165_alert_with_multiple_runbooks/pod_not_ready.yaml b/tests/llm/fixtures/test_ask_holmes/165_alert_with_multiple_runbooks/pod_not_ready.yaml new file mode 100644 index 0000000000..04fe7b11f1 --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/165_alert_with_multiple_runbooks/pod_not_ready.yaml @@ -0,0 +1,18 @@ +apiVersion: v1 +kind: Pod +metadata: + name: search-engine-service +spec: + containers: + - name: get-details + image: busybox + command: ["sh", "-c", "while true; do echo 'Running...'; sleep 5; done"] + readinessProbe: + exec: + command: + - sh + - -c + - "exit 1" + initialDelaySeconds: 5 + periodSeconds: 5 + failureThreshold: 3 diff --git a/tests/llm/fixtures/test_ask_holmes/165_alert_with_multiple_runbooks/runbook_catalog.json b/tests/llm/fixtures/test_ask_holmes/165_alert_with_multiple_runbooks/runbook_catalog.json new file mode 100644 index 0000000000..1279bff658 --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/165_alert_with_multiple_runbooks/runbook_catalog.json @@ -0,0 +1,23 @@ +[ + { + "id": "8fe8e24d-6b53-47a6-92a5-b389938fa823", + "symptom": "Pod is not ready", + "title": "Check probes", + "alerts": [ + "ProbeFailure" + ] + }, + { + "id": "40296b5c-2cb5-41df-b1f5-93441f894c44", + "symptom": "Pod continuously crashing and restarting", + "title": "Pod Crashlooping Debugging Runbook" + }, + { + "id": "b2ffd311-f339-46a2-b379-1e53eb0bc1ed", + "symptom": "", + "title": "Check relevant logs & manufests", + "alerts": [ + "ProbeFailure" + ] + } +] diff --git a/tests/llm/fixtures/test_ask_holmes/165_alert_with_multiple_runbooks/runbook_content_40296b5c-2cb5-41df-b1f5-93441f894c44.json b/tests/llm/fixtures/test_ask_holmes/165_alert_with_multiple_runbooks/runbook_content_40296b5c-2cb5-41df-b1f5-93441f894c44.json new file mode 100644 index 0000000000..74c64d6d9a --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/165_alert_with_multiple_runbooks/runbook_content_40296b5c-2cb5-41df-b1f5-93441f894c44.json @@ -0,0 +1,6 @@ +{ + "id": "40296b5c-2cb5-41df-b1f5-93441f894c44", + "symptom": "Pod continuously crashing and restarting", + "title": "Pod Crashlooping Debugging Runbook", + "instruction": "# Pod Crashlooping Debugging Runbook\n\n## Overview\nThis runbook provides step-by-step instructions for debugging pods that are crashlooping (continuously crashing and restarting).\n\n## Step 1: Check Pod Status\n```bash\nkubectl get pods -n \nkubectl describe pod -n \n```\n\nLook for:\n- Restart count\n- Current status (CrashLoopBackOff, Error, etc.)\n- Events section for error messages\n\n## Step 2: Check Pod Logs\n```bash\nkubectl logs -n \nkubectl logs -n --previous # Previous container logs\n```\n\nCommon issues to look for:\n- Application startup errors\n- Configuration file issues\n- Database connection failures\n- Missing environment variables\n\n## Step 3: Check Resource Constraints\n```bash\nkubectl top pod -n \nkubectl describe pod -n | grep -A 5 -B 5 \"Limits\\|Requests\"\n```\n\nCheck for:\n- Memory limits too low\n- CPU limits too restrictive\n- Resource quotas exceeded\n\n## Step 4: Check Configuration\n```bash\nkubectl get pod -n -o yaml\n```\n\nVerify:\n- Environment variables are set correctly\n- Volume mounts are working\n- ConfigMaps and Secrets are properly referenced\n\n## Step 5: Check Dependencies\n- Database connectivity\n- External service availability\n- Network policies blocking traffic\n- Service account permissions\n\n## Common Solutions\n1. **Memory Issues**: Increase memory limits or fix memory leaks\n2. **Configuration Issues**: Fix environment variables or config files\n3. **Dependency Issues**: Ensure external services are available\n4. **Image Issues**: Check if container image is correct and accessible\n\n## Prevention\n- Set appropriate resource requests and limits\n- Use health checks (liveness and readiness probes)\n- Implement proper error handling in applications\n- Monitor resource usage and set up alerts" +} diff --git a/tests/llm/fixtures/test_ask_holmes/165_alert_with_multiple_runbooks/runbook_content_8fe8e24d-6b53-47a6-92a5-b389938fa823.json b/tests/llm/fixtures/test_ask_holmes/165_alert_with_multiple_runbooks/runbook_content_8fe8e24d-6b53-47a6-92a5-b389938fa823.json new file mode 100644 index 0000000000..348f6d4aa5 --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/165_alert_with_multiple_runbooks/runbook_content_8fe8e24d-6b53-47a6-92a5-b389938fa823.json @@ -0,0 +1,9 @@ +{ + "id": "8fe8e24d-6b53-47a6-92a5-b389938fa823", + "symptom": "Pod is not ready", + "title": "Kafka Topic and Application Mapping", + "instruction": "Check pod template parameters such as: \n* pod priority\n* resources - maybe it tries to use unavailable resource, such as GPU but there is limited number of nodes with GPU\n* readiness and liveness probes may be incorrect - wrong port or command, check is failing too fast due to short timeout for response\n* stuck or long running init containers", + "alerts": [ + "ProbeFailure" + ] +} diff --git a/tests/llm/fixtures/test_ask_holmes/165_alert_with_multiple_runbooks/runbook_content_b2ffd311-f339-46a2-b379-1e53eb0bc1ed.json b/tests/llm/fixtures/test_ask_holmes/165_alert_with_multiple_runbooks/runbook_content_b2ffd311-f339-46a2-b379-1e53eb0bc1ed.json new file mode 100644 index 0000000000..3bd8bb1232 --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/165_alert_with_multiple_runbooks/runbook_content_b2ffd311-f339-46a2-b379-1e53eb0bc1ed.json @@ -0,0 +1,9 @@ +{ + "id": "b2ffd311-f339-46a2-b379-1e53eb0bc1ed", + "symptom": "Pod is crashlooping", + "title": "Signup Service Debugging Runbook", + "instruction": "Check template via kubectl -n $NAMESPACE get pod $POD.\nCheck pod events via kubectl -n $NAMESPACE describe pod $POD.\nCheck pod logs via kubectl -n $NAMESPACE logs $POD -c $CONTAINER", + "alerts": [ + "ProbeFailure" + ] +} diff --git a/tests/llm/fixtures/test_ask_holmes/165_alert_with_multiple_runbooks/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/165_alert_with_multiple_runbooks/test_case.yaml new file mode 100644 index 0000000000..2b567cad8b --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/165_alert_with_multiple_runbooks/test_case.yaml @@ -0,0 +1,16 @@ +# This test is used to test the ability of HolmesGPT to find the relevant runbook for a given alert +# TODO: Add check for the tool calls that actually checked that the relevant runbboks were fetched and used +user_prompt: "What is the issue with robusta alert 8604ba7a-2e54-4d1d-ae92-e1d1cfea7b98" +expected_output: + - ProbeFailure for pod search-engine-service is due to misconfigured readiness probe +before_test: | + kubectl apply -f ./pod_not_ready.yaml + sleep 10 +after_test: | + kubectl delete -f ./pod_not_ready.yaml +evaluation: + correctness: 1 +tags: + - kubernetes + - runbooks +test_type: cluster diff --git a/tests/llm/utils/mock_dal.py b/tests/llm/utils/mock_dal.py index dea07c5fd0..f067fa4478 100644 --- a/tests/llm/utils/mock_dal.py +++ b/tests/llm/utils/mock_dal.py @@ -7,7 +7,7 @@ from pydantic import TypeAdapter from holmes.core.supabase_dal import SupabaseDal, FindingType -from holmes.core.tool_calling_llm import ResourceInstructions +from holmes.core.resource_instruction import ResourceInstructions from holmes.plugins.runbooks import RobustaRunbookInstruction from holmes.utils.global_instructions import Instructions from tests.llm.utils.test_case_utils import read_file diff --git a/tests/llm/utils/test_case_utils.py b/tests/llm/utils/test_case_utils.py index 4a565c2ef0..0421cf38bb 100644 --- a/tests/llm/utils/test_case_utils.py +++ b/tests/llm/utils/test_case_utils.py @@ -8,6 +8,7 @@ import pytest from pydantic import BaseModel, TypeAdapter, ValidationError, ConfigDict +from holmes.core.resource_instruction import ResourceInstructions from tests.llm.utils.test_env_vars import ( MODEL, CLASSIFIER_MODEL, @@ -18,7 +19,6 @@ from holmes.config import Config from holmes.core.llm import DefaultLLM -from holmes.core.tool_calling_llm import ResourceInstructions from tests.llm.utils.constants import ALLOWED_EVAL_TAGS, get_allowed_tags_list diff --git a/tests/test_issue_investigator.py b/tests/test_issue_investigator.py index 9cc2b35d03..dcfd3d61fb 100644 --- a/tests/test_issue_investigator.py +++ b/tests/test_issue_investigator.py @@ -4,8 +4,6 @@ from holmes.config import Config from holmes.core.issue import Issue from holmes.core.models import InvestigateRequest -from holmes.core.resource_instruction import ResourceInstructionDocument -from holmes.core.tool_calling_llm import ResourceInstructions def _test_investigate_issue_using_fetch_webpage(): @@ -23,9 +21,6 @@ def _test_investigate_issue_using_fetch_webpage(): raw_data = investigate_request.model_dump() runbook_url = "https://containersolutions.github.io/runbooks/posts/kubernetes/create-container-error/" - resource_instructions = ResourceInstructions( - instructions=[], documents=[ResourceInstructionDocument(url=runbook_url)] - ) console = Console() config = Config.load_from_env() ai = config.create_issue_investigator(console) @@ -42,7 +37,6 @@ def _test_investigate_issue_using_fetch_webpage(): prompt=investigate_request.prompt_template, console=console, post_processing_prompt=HOLMES_POST_PROCESSING_PROMPT, - instructions=resource_instructions, ) webpage_tool_calls = list( @@ -69,7 +63,6 @@ def _test_investigate_issue_without_fetch_webpage(): prompt_template="builtin://generic_investigation.jinja2", ) raw_data = investigate_request.model_dump() - resource_instructions = ResourceInstructions(instructions=[], documents=[]) console = Console() config = Config.load_from_env() ai = config.create_issue_investigator(console) @@ -87,7 +80,6 @@ def _test_investigate_issue_without_fetch_webpage(): prompt=investigate_request.prompt_template, console=console, post_processing_prompt=HOLMES_POST_PROCESSING_PROMPT, - instructions=resource_instructions, ) webpage_tool_calls = list( From 31d2e2e70ed6a5a7ecb4ab9ed887e3df20d1eafc Mon Sep 17 00:00:00 2001 From: Natan Yellin Date: Sun, 21 Dec 2025 20:24:57 +0200 Subject: [PATCH 07/27] Add image link comments to docker workflows (#1210) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ## Summary - add automated comments to dev docker build workflow with links to built images - add automated comments to release docker build workflow with links to built images and tags - adjust workflow permissions to allow posting comments on PRs or commits ## Testing - not run (not needed for workflow changes) ------ [Codex Task](https://chatgpt.com/codex/tasks/task_b_6947d623f1a8832797c92848985d3af4) ## Summary by CodeRabbit * **Chores** * Updated CI/CD workflows to automatically post Docker image build details and registry links as comments on pull requests or commits. Comments provide full and short image tags along with registry information upon image release. ✏️ Tip: You can customize this high-level summary in your review settings. --------- Signed-off-by: Robusta Runner Signed-off-by: Mohse Morad Co-authored-by: moshemorad Signed-off-by: Filip Grebowski --- .github/workflows/docker-dev-images.yaml | 51 +++++++++++++++++++++++- 1 file changed, 50 insertions(+), 1 deletion(-) diff --git a/.github/workflows/docker-dev-images.yaml b/.github/workflows/docker-dev-images.yaml index 63e3927362..333bafcf94 100644 --- a/.github/workflows/docker-dev-images.yaml +++ b/.github/workflows/docker-dev-images.yaml @@ -10,8 +10,9 @@ jobs: runs-on: ubuntu-latest permissions: - contents: 'read' + contents: 'write' id-token: 'write' + pull-requests: 'write' steps: - uses: actions/checkout@v4 @@ -53,3 +54,51 @@ jobs: echo "Docker image pushed to:" echo " us-central1-docker.pkg.dev/robusta-development/temporary-builds/holmes:${{ github.sha }}" echo " us-central1-docker.pkg.dev/robusta-development/temporary-builds/holmes:${{ steps.short_sha.outputs.sha_short }}" + + - name: Comment with image details + uses: actions/github-script@v7 + with: + script: | + const owner = context.repo.owner; + const repo = context.repo.repo; + const branch = context.ref.replace('refs/heads/', ''); + const sha = context.sha; + const shortSha = `${{ steps.short_sha.outputs.sha_short }}`; + + const fullTag = `us-central1-docker.pkg.dev/robusta-development/temporary-builds/holmes:${sha}`; + const shortTag = `us-central1-docker.pkg.dev/robusta-development/temporary-builds/holmes:${shortSha}`; + const registryLink = 'https://console.cloud.google.com/artifacts/docker/robusta-development/us-central1/temporary-builds/holmes?project=robusta-development'; + + const message = [ + 'Dev Docker images are ready for this commit:', + '', + `- [${fullTag}](${registryLink})`, + `- [${shortTag}](${registryLink})`, + '', + 'Use either tag to pull the image for testing.' + ].join('\n'); + + const prs = await github.paginate(github.rest.pulls.list, { + owner, + repo, + state: 'open', + head: `${owner}:${branch}`, + }); + + if (prs.length > 0) { + await github.rest.issues.createComment({ + owner, + repo, + issue_number: prs[0].number, + body: message, + }); + core.info(`Commented on PR #${prs[0].number}`); + } else { + await github.rest.repos.createCommitComment({ + owner, + repo, + commit_sha: sha, + body: message, + }); + core.info('Commented on commit'); + } From 89f16c1ee611f954126b16e0e890c633953534d0 Mon Sep 17 00:00:00 2001 From: Tomer Date: Mon, 22 Dec 2025 21:18:47 +0200 Subject: [PATCH 08/27] verify setup success in eval 64_keda_vs_hpa_confusio and change to hard (#1223) Signed-off-by: Tomer Keshet Signed-off-by: Filip Grebowski --- .../64_keda_vs_hpa_confusion/manifest.yaml | 6 ++--- .../64_keda_vs_hpa_confusion/test_case.yaml | 24 ++++++++++++++++--- 2 files changed, 24 insertions(+), 6 deletions(-) diff --git a/tests/llm/fixtures/test_ask_holmes/64_keda_vs_hpa_confusion/manifest.yaml b/tests/llm/fixtures/test_ask_holmes/64_keda_vs_hpa_confusion/manifest.yaml index 9cf4dc4335..9c8e705f68 100644 --- a/tests/llm/fixtures/test_ask_holmes/64_keda_vs_hpa_confusion/manifest.yaml +++ b/tests/llm/fixtures/test_ask_holmes/64_keda_vs_hpa_confusion/manifest.yaml @@ -1,13 +1,13 @@ apiVersion: v1 kind: Namespace metadata: - name: invoices-64 + name: app-64 --- apiVersion: apps/v1 kind: Deployment metadata: name: invoice-generator - namespace: invoices-64 + namespace: app-64 labels: app: invoice-generator version: v1.2.0 @@ -46,7 +46,7 @@ apiVersion: autoscaling/v2 kind: HorizontalPodAutoscaler metadata: name: invoice-generator - namespace: invoices-64 + namespace: app-64 spec: scaleTargetRef: apiVersion: apps/v1 diff --git a/tests/llm/fixtures/test_ask_holmes/64_keda_vs_hpa_confusion/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/64_keda_vs_hpa_confusion/test_case.yaml index 281e6cb0f4..1b11b2aee5 100644 --- a/tests/llm/fixtures/test_ask_holmes/64_keda_vs_hpa_confusion/test_case.yaml +++ b/tests/llm/fixtures/test_ask_holmes/64_keda_vs_hpa_confusion/test_case.yaml @@ -4,8 +4,26 @@ expected_output: | - KEDA is not installed in your cluster OR you have no KEDA scaler defined for this resource (either answer is fine) - But I found an HPA attached to the invoice-generator deployment # - Current CPU usage is 9%, which is below the 60% threshold configured in the HPA. That's why it isn't scaling up -generate_mocks: False + tags: - - medium -before_test: 'kubectl apply -f ./manifest.yaml' + - hard +before_test: | + kubectl apply -f ./manifest.yaml + # Wait for invoice-generator pod with logs (60s total) - MUST succeed or test fails + LOG_READY=false + for i in {1..20}; do + if kubectl wait --for=condition=ready pod -l app=invoice-generator -n app-64 --timeout=1s 2>/dev/null && kubectl logs -l app=invoice-generator -n app-64 --tail=10 2>/dev/null | grep -q "Generating invoice"; then + echo "✅ invoice-generator pod ready with logs!" + LOG_READY=true + break + else + echo "⏳ Attempt $i/20: pod/logs not ready, waiting 3s..." + sleep 3 + fi + done + if [ "$LOG_READY" = false ]; then + echo "❌ Pod/logs failed after 60s" + kubectl get pods -l app=invoice-generator -n app-64 + exit 1 + fi after_test: 'kubectl delete -f ./manifest.yaml' From bf8bd69f4f1189ff68a196d8dc177249f26b514c Mon Sep 17 00:00:00 2001 From: Tomer Date: Tue, 23 Dec 2025 09:06:00 +0200 Subject: [PATCH 09/27] 57_wrong_namespace fix judge to allow get logs from the right namespace (#1222) Signed-off-by: Tomer Keshet Signed-off-by: Filip Grebowski --- .../57_wrong_namespace/test_case.yaml | 29 +++++++++++++++---- 1 file changed, 24 insertions(+), 5 deletions(-) diff --git a/tests/llm/fixtures/test_ask_holmes/57_wrong_namespace/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/57_wrong_namespace/test_case.yaml index 4d36270a67..78b7c89fa1 100644 --- a/tests/llm/fixtures/test_ask_holmes/57_wrong_namespace/test_case.yaml +++ b/tests/llm/fixtures/test_ask_holmes/57_wrong_namespace/test_case.yaml @@ -1,11 +1,30 @@ # we ask about app-57 namespace but the deployment is in test-57 namespace user_prompt: "get last logs from video-streamer in app-57 namespace" -expected_output: "video-streamer not found in namespace app-57, did you mean namespace test-57?" -before_test: 'kubectl apply -f ./manifest.yaml' +expected_output: + - "MUST explicitly state that video-streamer was not found in app-57 namespace OR explicitly mention the namespace mismatch between app-57 and test-57" + - "MUST successfully locate video-streamer in test-57 AND either retrieve the logs OR ask if user wants logs from test-57" +before_test: | + kubectl apply -f ./manifest.yaml + + # Wait for video-streamer pod to be ready (60 seconds total) - MUST succeed or test fails + POD_READY=false + for i in {1..12}; do + if kubectl wait --for=condition=ready pod -l app=video-streamer -n test-57 --timeout=5s 2>/dev/null; then + echo "✅ video-streamer pod is ready!" + POD_READY=true + break + else + echo "⏳ Attempt $i/12: video-streamer pod not ready yet, waiting 5s..." + sleep 5 + fi + done + + if [ "$POD_READY" = false ]; then + echo "❌ video-streamer pod failed to become ready after 60 seconds" + kubectl get pods -l app=video-streamer -n test-57 # Show pod status for debugging + exit 1 + fi after_test: 'kubectl delete -f ./manifest.yaml' -evaluation: - correctness: 0 -generate_mocks: False tags: - logs - medium From fa7381931b3d45a56cd51614df1037dad66fcff1 Mon Sep 17 00:00:00 2001 From: Tomer Date: Tue, 23 Dec 2025 09:52:46 +0200 Subject: [PATCH 10/27] 82_pod_anti_affinity_conflict: ensure anti-affinity rule exist in setup (#1221) Signed-off-by: Tomer Keshet Signed-off-by: Filip Grebowski --- .../manifest.yaml | 6 ++--- .../test_case.yaml | 22 ++++++++++++++++--- 2 files changed, 22 insertions(+), 6 deletions(-) diff --git a/tests/llm/fixtures/test_ask_holmes/82_pod_anti_affinity_conflict/manifest.yaml b/tests/llm/fixtures/test_ask_holmes/82_pod_anti_affinity_conflict/manifest.yaml index bef1b37d72..e63d86578b 100644 --- a/tests/llm/fixtures/test_ask_holmes/82_pod_anti_affinity_conflict/manifest.yaml +++ b/tests/llm/fixtures/test_ask_holmes/82_pod_anti_affinity_conflict/manifest.yaml @@ -1,16 +1,16 @@ apiVersion: v1 kind: Namespace metadata: - name: namespace-82 + name: app-82 --- # Deployment with strict pod anti-affinity apiVersion: apps/v1 kind: Deployment metadata: name: silver-puffin - namespace: namespace-82 + namespace: app-82 spec: - replicas: 8 # Trying to schedule 8 pods with anti-affinity (enough to demonstrate the issue) + replicas: 20 # Trying to more pods than nodes selector: matchLabels: app: silver-puffin diff --git a/tests/llm/fixtures/test_ask_holmes/82_pod_anti_affinity_conflict/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/82_pod_anti_affinity_conflict/test_case.yaml index 52a361b319..af2cb9e9ab 100644 --- a/tests/llm/fixtures/test_ask_holmes/82_pod_anti_affinity_conflict/test_case.yaml +++ b/tests/llm/fixtures/test_ask_holmes/82_pod_anti_affinity_conflict/test_case.yaml @@ -1,12 +1,28 @@ -user_prompt: "Why can't the silver-puffin pods be scheduled in namespace-82?" +user_prompt: "Why can't the silver-puffin pods be scheduled in app-82?" tags: - kubernetes - medium expected_output: - Silver-puffin pods cannot be scheduled because pod anti-affinity rules require them on different nodes but insufficient nodes available - - It's ok if other reasons are mentioned too as long as the anti-affinity conflict is clearly stated before_test: | kubectl apply -f ./manifest.yaml - sleep 20 + + # Wait for anti-affinity scheduling conflict (60 seconds total) + CONFLICT_DETECTED=false + for i in {1..12}; do + if kubectl get events -n app-82 --no-headers 2>/dev/null | grep "silver-puffin" | grep -q "didn't match pod anti-affinity rules"; then + echo "✅ Anti-affinity conflict detected!" + CONFLICT_DETECTED=true + break + else + echo "⏳ Attempt $i/12: Waiting for anti-affinity conflict event..." + sleep 5 + fi + done + + if [ "$CONFLICT_DETECTED" = false ]; then + echo "❌ Anti-affinity conflict not detected after 60 seconds" + exit 1 + fi after_test: | kubectl delete -f ./manifest.yaml From ea265593f79206e576da06e8bd37c2599e845066 Mon Sep 17 00:00:00 2001 From: Tomer Date: Tue, 23 Dec 2025 12:38:21 +0200 Subject: [PATCH 11/27] improving several tests setup and instructions (#1227) Signed-off-by: Tomer Keshet Signed-off-by: Filip Grebowski --- .../108_logs_nearby_lines/manifest.yaml | 4 +-- .../108_logs_nearby_lines/test_case.yaml | 31 +++++++++++++++++-- .../16_failed_no_toolset_found/test_case.yaml | 2 +- .../17_oom_kill/test_case.yaml | 21 +++++++++++-- .../manifests/namespace.yaml | 4 --- .../test_case.yaml | 10 +++++- 6 files changed, 60 insertions(+), 12 deletions(-) delete mode 100644 tests/llm/fixtures/test_ask_holmes/23_app_error_in_current_logs/manifests/namespace.yaml diff --git a/tests/llm/fixtures/test_ask_holmes/108_logs_nearby_lines/manifest.yaml b/tests/llm/fixtures/test_ask_holmes/108_logs_nearby_lines/manifest.yaml index b88ad00fdc..53b9d647bb 100644 --- a/tests/llm/fixtures/test_ask_holmes/108_logs_nearby_lines/manifest.yaml +++ b/tests/llm/fixtures/test_ask_holmes/108_logs_nearby_lines/manifest.yaml @@ -77,8 +77,8 @@ stringData: echo "2024-01-15T10:07:$(printf '%02d' $i).200Z [HTTP] 500 GET /api/health - Connection refused" done - # Keep pod running - sleep 3600 + # Keep pod running indefinitely + while true; do sleep 30; done --- apiVersion: apps/v1 diff --git a/tests/llm/fixtures/test_ask_holmes/108_logs_nearby_lines/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/108_logs_nearby_lines/test_case.yaml index 1d14346743..749fa5187f 100644 --- a/tests/llm/fixtures/test_ask_holmes/108_logs_nearby_lines/test_case.yaml +++ b/tests/llm/fixtures/test_ask_holmes/108_logs_nearby_lines/test_case.yaml @@ -3,7 +3,7 @@ description: | The cause is a configuration change To understand what happened, holmes needs to see logs immediate BEFORE the first occurrence -user_prompt: "What is causing connection refused errors in api-service namespace app-108?" +user_prompt: "What is causing connection refused errors in api-service in namespace app-108?" tags: - logs @@ -12,7 +12,34 @@ tags: before_test: | kubectl create namespace app-108 || true kubectl apply -f ./manifest.yaml -n app-108 - kubectl wait --for=condition=ready pod -l app=api-service -n app-108 --timeout=60s || true + # Wait for pod to be ready and specific log lines to appear (60s total) - MUST succeed or test fails + LOGS_READY=false + for i in {1..20}; do + # First check if pod exists and is ready + if kubectl wait --for=condition=ready pod -l app=api-service -n app-108 --timeout=1s 2>/dev/null; then + # Pod is ready, now get its name and check for specific log lines + POD_NAME=$(kubectl get pods -n app-108 -l app=api-service -o jsonpath='{.items[0].metadata.name}' 2>/dev/null) + if [ -n "$POD_NAME" ] && kubectl logs "$POD_NAME" -n app-108 2>/dev/null | grep -q "2024-01-15T10:05:00.000Z \[WARN\] Configuration file change detected" && kubectl logs "$POD_NAME" -n app-108 2>/dev/null | grep -q "2024-01-15T10:07:30.200Z \[HTTP\] 500 GET /api/health - Connection refused"; then + echo "✅ Required log lines detected!" + LOGS_READY=true + break + else + echo "⏳ Attempt $i/20: pod ready but specific log lines not found yet, checking in 3s..." + fi + else + echo "⏳ Attempt $i/20: waiting for pod to be ready, checking in 3s..." + fi + sleep 3 + done + if [ "$LOGS_READY" = false ]; then + echo "❌ Required log lines not found after 60s" + POD_NAME=$(kubectl get pods -n app-108 -l app=api-service -o jsonpath='{.items[0].metadata.name}' 2>/dev/null) + if [ -n "$POD_NAME" ]; then + kubectl logs "$POD_NAME" -n app-108 --tail=50 + fi + kubectl get pods -n app-108 + exit 1 + fi after_test: | kubectl delete namespace app-108 || true diff --git a/tests/llm/fixtures/test_ask_holmes/16_failed_no_toolset_found/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/16_failed_no_toolset_found/test_case.yaml index c20b647b00..517e731f08 100644 --- a/tests/llm/fixtures/test_ask_holmes/16_failed_no_toolset_found/test_case.yaml +++ b/tests/llm/fixtures/test_ask_holmes/16_failed_no_toolset_found/test_case.yaml @@ -1,4 +1,4 @@ -user_prompt: "Why is my Digital ocean droplet not working fine" +user_prompt: "Why is my Digital Ocean droplet 'web-server-prod-01' not working fine. Is it down?" expected_output: - The answer should indicate Holmes doesn't have access to Digital Ocean - The answer should include a link to https://holmesgpt.dev/data-sources/custom-toolsets/ or https://holmesgpt.dev/data-sources/remote-mcp-servers/ diff --git a/tests/llm/fixtures/test_ask_holmes/17_oom_kill/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/17_oom_kill/test_case.yaml index 85588dcfd7..398b22bc56 100644 --- a/tests/llm/fixtures/test_ask_holmes/17_oom_kill/test_case.yaml +++ b/tests/llm/fixtures/test_ask_holmes/17_oom_kill/test_case.yaml @@ -1,10 +1,27 @@ -user_prompt: "What is the issue with analytics-exporter-fast?" +user_prompt: "What is the issue with analytics-exporter-fast in namespace app-17?" expected_output: - The result mentions analytics-exporter-fast is getting OOMKILLED before_test: | kubectl create namespace app-17 || true kubectl apply -f ./fast_oom_deployment.yaml -n app-17 - sleep 5 + # Wait for OOM kill to occur (60s total) - MUST succeed or test fails + OOM_DETECTED=false + for i in {1..20}; do + if kubectl get pods -n app-17 -o jsonpath='{.items[*].status.containerStatuses[*].lastState.terminated.reason}' 2>/dev/null | grep -q "OOMKilled"; then + echo "✅ OOM kill detected!" + OOM_DETECTED=true + break + else + echo "⏳ Attempt $i/20: waiting for OOM kill, checking in 3s..." + sleep 3 + fi + done + if [ "$OOM_DETECTED" = false ]; then + echo "❌ No OOM kill detected after 60s" + kubectl get pods -n app-17 + kubectl get events -n app-17 + exit 1 + fi after_test: | kubectl delete -f ./fast_oom_deployment.yaml -n app-17 kubectl delete namespace app-17 || true diff --git a/tests/llm/fixtures/test_ask_holmes/23_app_error_in_current_logs/manifests/namespace.yaml b/tests/llm/fixtures/test_ask_holmes/23_app_error_in_current_logs/manifests/namespace.yaml deleted file mode 100644 index cd7488548a..0000000000 --- a/tests/llm/fixtures/test_ask_holmes/23_app_error_in_current_logs/manifests/namespace.yaml +++ /dev/null @@ -1,4 +0,0 @@ -apiVersion: v1 -kind: Namespace -metadata: - name: app-23 diff --git a/tests/llm/fixtures/test_ask_holmes/23_app_error_in_current_logs/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/23_app_error_in_current_logs/test_case.yaml index 35b87db750..e54dbb559b 100644 --- a/tests/llm/fixtures/test_ask_holmes/23_app_error_in_current_logs/test_case.yaml +++ b/tests/llm/fixtures/test_ask_holmes/23_app_error_in_current_logs/test_case.yaml @@ -11,16 +11,24 @@ before_test: | # Wait for curl pod to generate some error logs (replaces fixed 30s sleep) echo "Waiting for curl pod to generate DNS error logs..." + ERROR_FOUND=false for i in {1..30}; do - if kubectl logs -n app-23 -l app=curl-app --tail=5 2>/dev/null | grep -q "Failed to reach"; then + if kubectl logs -n app-23 -l app=curl-app 2>/dev/null | grep -q "Failed to reach"; then echo "✅ Error logs detected after $i seconds" + ERROR_FOUND=true break fi echo "⏳ Attempt $i/30: Waiting for error logs..." sleep 1 done + if [ "$ERROR_FOUND" = false ]; then + echo "❌ No error logs found after 30 seconds" + kubectl logs -n app-23 -l app=curl-app + exit 1 + fi after_test: | kubectl delete -f ./manifests/ + kubectl delete namespace app-23 || true tags: - medium - kubernetes From e0d4c99a082e96f70f4aef67775cbcaae307e223 Mon Sep 17 00:00:00 2001 From: Tomer Date: Tue, 23 Dec 2025 17:44:52 +0200 Subject: [PATCH 12/27] stabilizing more tests (#1233) Signed-off-by: Tomer Keshet Signed-off-by: Filip Grebowski --- .../12_job_crashing/kubectl_describe_job.txt | 40 ------------------- .../12_job_crashing/kubectl_describe_pod.txt | 40 ------------------- .../kubectl_find_resource_job.txt | 6 --- .../kubectl_find_resource_pod.txt | 3 -- .../12_job_crashing/kubectl_get.txt | 7 ---- .../kubectl_get_by_kind_in_namespace.txt | 40 ------------------- .../12_job_crashing/kubectl_get_yaml.txt | 40 ------------------- .../kubectl_lineage_children.txt | 9 ----- .../12_job_crashing/kubectl_logs.txt | 10 ----- .../12_job_crashing/test_case.yaml | 24 ++++++++++- .../73a_time_window_anomaly/generate_logs.py | 11 +++++ .../73a_time_window_anomaly/manifest.yaml | 2 +- .../73a_time_window_anomaly/test_case.yaml | 39 +++++++++++++++--- .../73b_time_window_anomaly/generate_logs.py | 11 +++++ .../73b_time_window_anomaly/manifest.yaml | 2 +- .../73b_time_window_anomaly/test_case.yaml | 39 +++++++++++++++--- 16 files changed, 113 insertions(+), 210 deletions(-) delete mode 100644 tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_describe_job.txt delete mode 100644 tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_describe_pod.txt delete mode 100644 tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_find_resource_job.txt delete mode 100644 tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_find_resource_pod.txt delete mode 100644 tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_get.txt delete mode 100644 tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_get_by_kind_in_namespace.txt delete mode 100644 tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_get_yaml.txt delete mode 100644 tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_lineage_children.txt delete mode 100644 tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_logs.txt diff --git a/tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_describe_job.txt b/tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_describe_job.txt deleted file mode 100644 index c61a82147d..0000000000 --- a/tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_describe_job.txt +++ /dev/null @@ -1,40 +0,0 @@ -{"toolset_name": "kubernetes/core", "tool_name": "kubectl_describe", "match_params": {"kind": "job", "name": "java-api-checker", "namespace": "default"}} -{"schema_version": "robusta:v1.0.0", "status": "success", "error": null, "return_code": 0, "url": null, "invocation": "kubectl_describe command", "params": {"kind": "job", "name": "java-api-checker", "namespace": "default"}} -stdout: -Name: java-api-checker -Namespace: default -Selector: batch.kubernetes.io/controller-uid=b527d2de-04c0-4b36-962b-90b9e45724cd -Labels: batch.kubernetes.io/controller-uid=b527d2de-04c0-4b36-962b-90b9e45724cd - batch.kubernetes.io/job-name=java-api-checker - controller-uid=b527d2de-04c0-4b36-962b-90b9e45724cd - job-name=java-api-checker -Annotations: -Parallelism: 1 -Completions: 1 -Completion Mode: NonIndexed -Suspend: false -Backoff Limit: 1 -Start Time: Tue, 26 Nov 2024 15:16:52 +0100 -Pods Statuses: 1 Active (1 Ready) / 0 Succeeded / 0 Failed -Pod Template: - Labels: batch.kubernetes.io/controller-uid=b527d2de-04c0-4b36-962b-90b9e45724cd - batch.kubernetes.io/job-name=java-api-checker - controller-uid=b527d2de-04c0-4b36-962b-90b9e45724cd - job-name=java-api-checker - Containers: - java-beans: - Image: java-api-checker - Port: - Host Port: - Command: start.sh - Environment: - Mounts: - Volumes: - Node-Selectors: - Tolerations: -Events: - Type Reason Age From Message - ---- ------ ---- ---- ------- - Normal SuccessfulCreate 47s job-controller Created pod: java-api-checker-mdr44 - -stderr: diff --git a/tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_describe_pod.txt b/tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_describe_pod.txt deleted file mode 100644 index e840637c6d..0000000000 --- a/tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_describe_pod.txt +++ /dev/null @@ -1,40 +0,0 @@ -{"toolset_name": "kubernetes/core", "tool_name": "kubectl_describe", "match_params": {"kind": "pod", "name": "java-api-checker-mdr44", "namespace": "default"}} -{"schema_version": "robusta:v1.0.0", "status": "success", "error": null, "return_code": 0, "url": null, "invocation": "kubectl_describe command", "params": {"kind": "pod", "name": "java-api-checker-mdr44", "namespace": "default"}} -stdout: -Name: java-api-checker-mdr44 -Namespace: default -Selector: batch.kubernetes.io/controller-uid=b527d2de-04c0-4b36-962b-90b9e45724cd -Labels: batch.kubernetes.io/controller-uid=b527d2de-04c0-4b36-962b-90b9e45724cd - batch.kubernetes.io/job-name=java-api-checker - controller-uid=b527d2de-04c0-4b36-962b-90b9e45724cd - job-name=java-api-checker -Annotations: -Parallelism: 1 -Completions: 1 -Completion Mode: NonIndexed -Suspend: false -Backoff Limit: 1 -Start Time: Tue, 26 Nov 2024 15:16:52 +0100 -Pods Statuses: 1 Active (1 Ready) / 0 Succeeded / 0 Failed -Pod Template: - Labels: batch.kubernetes.io/controller-uid=b527d2de-04c0-4b36-962b-90b9e45724cd - batch.kubernetes.io/job-name=java-api-checker - controller-uid=b527d2de-04c0-4b36-962b-90b9e45724cd - job-name=java-api-checker - Containers: - java-beans: - Image: java-api-checker - Port: - Host Port: - Command: start.sh - Environment: - Mounts: - Volumes: - Node-Selectors: - Tolerations: -Events: - Type Reason Age From Message - ---- ------ ---- ---- ------- - Normal SuccessfulCreate 47s job-controller Created pod: java-api-checker-mdr44 - -stderr: diff --git a/tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_find_resource_job.txt b/tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_find_resource_job.txt deleted file mode 100644 index 3dfef75273..0000000000 --- a/tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_find_resource_job.txt +++ /dev/null @@ -1,6 +0,0 @@ -{"toolset_name": "kubernetes/core", "tool_name": "kubectl_find_resource", "match_params": {"kind": "job", "keyword": "java-api-checker"}} -{"schema_version": "robusta:v1.0.0", "status": "success", "error": null, "return_code": 0, "url": null, "invocation": "kubectl_find_resource command", "params": {"kind": "job", "keyword": "java-api-checker"}} -stdout: -default java-api-checker Running 0/1 45s 45s java-beans busybox batch.kubernetes.io/controller-uid=b527d2de-04c0-4b36-962b-90b9e45724cd batch.kubernetes.io/controller-uid=b527d2de-04c0-4b36-962b-90b9e45724cd,batch.kubernetes.io/job-name=java-api-checker,controller-uid=b527d2de-04c0-4b36-962b-90b9e45724cd,job-name=java-api-checker - -stderr: diff --git a/tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_find_resource_pod.txt b/tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_find_resource_pod.txt deleted file mode 100644 index 04ce7717be..0000000000 --- a/tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_find_resource_pod.txt +++ /dev/null @@ -1,3 +0,0 @@ -{"toolset_name": "kubernetes/core", "tool_name": "kubectl_find_resource", "match_params": {"kind": "pod", "keyword": "java-api-checker"}} -{"schema_version": "robusta:v1.0.0", "status": "success", "error": null, "return_code": 0, "url": null, "invocation": "kubectl_find_resource command", "params": {"kind": "pod", "keyword": "java-api-checker"}} -java-api-checker-mdr44 1/1 Running 0 48s 10.244.0.92 kind-control-plane diff --git a/tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_get.txt b/tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_get.txt deleted file mode 100644 index 486bb894ca..0000000000 --- a/tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_get.txt +++ /dev/null @@ -1,7 +0,0 @@ -{"toolset_name": "kubernetes/core", "tool_name": "kubectl_get_by_name", "match_params": {"kind": "pod", "name": "java-api-checker-mdr44", "namespace": "default"}} -{"schema_version": "robusta:v1.0.0", "status": "success", "error": null, "return_code": 0, "url": null, "invocation": "kubectl_get_by_name command", "params": {"kind": "pod", "name": "java-api-checker-mdr44", "namespace": "default"}} -stdout: -NAME READY STATUS RESTARTS AGE IP NODE NOMINATED NODE READINESS GATES LABELS -java-api-checker-mdr44 1/1 Running 0 48s 10.244.0.92 kind-control-plane batch.kubernetes.io/controller-uid=b527d2de-04c0-4b36-962b-90b9e45724cd,batch.kubernetes.io/job-name=java-api-checker,controller-uid=b527d2de-04c0-4b36-962b-90b9e45724cd,job-name=java-api-checker - -stderr: diff --git a/tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_get_by_kind_in_namespace.txt b/tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_get_by_kind_in_namespace.txt deleted file mode 100644 index 0b02a06a88..0000000000 --- a/tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_get_by_kind_in_namespace.txt +++ /dev/null @@ -1,40 +0,0 @@ -{"toolset_name": "kubernetes/core", "tool_name": "kubectl_get_by_kind_in_namespace", "match_params": {"kind": "pod", "namespace": "default"}} -{"schema_version": "robusta:v1.0.0", "status": "success", "error": null, "return_code": 0, "url": null, "invocation": "kubectl_get_by_kind_in_namespace command", "params": {"kind": "pod", "namespace": "default"}} -stdout: -NAME READY STATUS RESTARTS AGE -alertmanager-robusta-kube-prometheus-st-alertmanager-0 2/2 Running 10 (3h51m ago) 6d22h -analytics-exporter-slow-684486cfb7-2b6lf 1/1 Running 68 (56m ago) 10d -curl-deployment-6c67b4656-qdzgg 1/1 Running 24 (3h51m ago) 31d -customer-orders-6f5cbdf85-c5fsf 2/2 Running 46 (3h51m ago) 29d -db-certs-authenticator-757f89d977-4qfst 0/1 CrashLoopBackOff 2121 (3m17s ago) 31d -frontend-service 1/1 Running 7 (3h51m ago) 10d -get-data-wpsss 0/1 Error 0 32d -grafana-k8s-monitoring-alloy-0 2/2 Running 126 (3h51m ago) 77d -grafana-k8s-monitoring-alloy-events-799cc88c88-9pllx 2/2 Running 131 (3h51m ago) 77d -grafana-k8s-monitoring-alloy-logs-t4ktj 2/2 Running 131 (3h51m ago) 77d -grafana-k8s-monitoring-kepler-2hntq 1/1 Running 46 (3h51m ago) 77d -grafana-k8s-monitoring-kube-state-metrics-5ff5b4947b-b7l57 1/1 Running 129 (3h51m ago) 77d -grafana-k8s-monitoring-opencost-5b67d55db5-nwp7l 1/1 Running 73 (3h51m ago) 77d -grafana-k8s-monitoring-prometheus-node-exporter-wjb7d 1/1 Running 46 (3h51m ago) 77d -java-api-checker-mdr44 1/1 Running 0 48s -kafka-client 0/1 Unknown 0 76d -kafka-consumer 0/1 Unknown 0 75d -kafka-controller-0 1/1 Running 110 (101m ago) 75d -kafka-controller-1 1/1 Running 114 (46m ago) 75d -kafka-controller-2 1/1 Running 118 (113m ago) 75d -kafka-producer 0/1 Unknown 0 75d -logging-agent 0/1 Init:CrashLoopBackOff 2510 (18s ago) 29d -login-app-58995d8584-pbv8p 3/3 Running 21 (3h51m ago) 10d -meme-deployment-74db7bc95c-gdgfg 1/1 Running 24 (3h51m ago) 31d -meme-deployment-74db7bc95c-qn84d 1/1 Running 24 (3h51m ago) 31d -prometheus-robusta-kube-prometheus-st-prometheus-0 2/2 Running 10 (3h51m ago) 6d22h -robusta-forwarder-5c5fdbbf57-8rzzh 1/1 Running 5 (3h51m ago) 6d22h -robusta-grafana-8588b8fb85-4x2zv 3/3 Running 15 (3h51m ago) 6d22h -robusta-holmes-84c7574786-mf5dq 1/1 Running 1 (3h51m ago) 19h -robusta-kube-prometheus-st-operator-6885c8f675-bbm7t 1/1 Running 8 (3h51m ago) 6d22h -robusta-kube-state-metrics-8667fd9775-lb5nx 1/1 Running 8 (3h51m ago) 6d22h -robusta-prometheus-node-exporter-dcnp5 1/1 Running 5 (3h51m ago) 6d22h -robusta-runner-86d77844dc-4xlwl 1/1 Running 6 (3h51m ago) 5d21h -user-profile-resources-659d4dd659-cq4kq 0/1 Pending 0 31d - -stderr: diff --git a/tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_get_yaml.txt b/tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_get_yaml.txt deleted file mode 100644 index 312f83f3b2..0000000000 --- a/tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_get_yaml.txt +++ /dev/null @@ -1,40 +0,0 @@ -{"toolset_name": "kubernetes/core", "tool_name": "kubectl_get_yaml", "match_params": {"kind": "pod", "name": "java-api-checker-mdr44", "namespace": "default"}} -{"schema_version": "robusta:v1.0.0", "status": "success", "error": null, "return_code": 0, "url": null, "invocation": "kubectl_get_yaml command", "params": {"kind": "pod", "name": "java-api-checker-mdr44", "namespace": "default"}} -stdout: -Name: java-api-checker-mdr44 -Namespace: default -Selector: batch.kubernetes.io/controller-uid=b527d2de-04c0-4b36-962b-90b9e45724cd -Labels: batch.kubernetes.io/controller-uid=b527d2de-04c0-4b36-962b-90b9e45724cd - batch.kubernetes.io/job-name=java-api-checker - controller-uid=b527d2de-04c0-4b36-962b-90b9e45724cd - job-name=java-api-checker -Annotations: -Parallelism: 1 -Completions: 1 -Completion Mode: NonIndexed -Suspend: false -Backoff Limit: 1 -Start Time: Tue, 26 Nov 2024 15:16:52 +0100 -Pods Statuses: 1 Active (1 Ready) / 0 Succeeded / 0 Failed -Pod Template: - Labels: batch.kubernetes.io/controller-uid=b527d2de-04c0-4b36-962b-90b9e45724cd - batch.kubernetes.io/job-name=java-api-checker - controller-uid=b527d2de-04c0-4b36-962b-90b9e45724cd - job-name=java-api-checker - Containers: - java-beans: - Image: java-api-checker - Port: - Host Port: - Command: start.sh - Environment: - Mounts: - Volumes: - Node-Selectors: - Tolerations: -Events: - Type Reason Age From Message - ---- ------ ---- ---- ------- - Normal SuccessfulCreate 47s job-controller Created pod: java-api-checker-mdr44 - -stderr: diff --git a/tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_lineage_children.txt b/tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_lineage_children.txt deleted file mode 100644 index 17fe756271..0000000000 --- a/tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_lineage_children.txt +++ /dev/null @@ -1,9 +0,0 @@ -{"toolset_name": "kubernetes/kube-lineage-extras", "tool_name": "kubectl_lineage_children", "match_params": {"kind": "job", "name": "java-api-checker", "namespace": "default"}} -{"schema_version": "robusta:v1.0.0", "status": "success", "error": null, "return_code": 0, "url": null, "invocation": "kubectl_lineage_children command", "params": {"kind": "job", "name": "java-api-checker", "namespace": "default"}} -stdout: -NAME READY STATUS AGE -Job/java-api-checker - 28d -├── Pod/java-api-checker-mdr44 0/1 Error 28d -│ └── Service/kubernetes - 77d - -stderr: diff --git a/tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_logs.txt b/tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_logs.txt deleted file mode 100644 index 14d0a89da1..0000000000 --- a/tests/llm/fixtures/test_ask_holmes/12_job_crashing/kubectl_logs.txt +++ /dev/null @@ -1,10 +0,0 @@ -{"toolset_name": "kubernetes/logs", "tool_name": "fetch_pod_logs", "match_params": {"pod_name": "java-api-checker-mdr44", "namespace": "default"}} -{"schema_version": "robusta:v1.0.0", "status": "success", "error": null, "return_code": 0, "url": null, "invocation": "fetch_pod_logs command", "params": {"pod_name": "java-api-checker-mdr44", "namespace": "default"}} -stdout: -Java Network Exception: -All host(s) tried for db query failed (tried: prod-db:3333) - no available connection and the queue has reached its max size 256 -All host(s) tried for db query failed (tried: prod-db:3333) - no available connection and the queue has reached its max size 256 -All host(s) tried for db query failed (tried: prod-db:3333) - no available connection and the queue has reached its max size 256 -All host(s) tried for db query failed (tried: prod-db:3333) - no available connection and the queue has reached its max size 256 - -stderr: diff --git a/tests/llm/fixtures/test_ask_holmes/12_job_crashing/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/12_job_crashing/test_case.yaml index dcf8a68c52..26cf52be66 100644 --- a/tests/llm/fixtures/test_ask_holmes/12_job_crashing/test_case.yaml +++ b/tests/llm/fixtures/test_ask_holmes/12_job_crashing/test_case.yaml @@ -8,7 +8,29 @@ before_test: | --from-file=generate_logs.py=./generate_logs.py \ -n app-12 --dry-run=client -o yaml | kubectl apply -f - kubectl apply -f ./job.yaml - sleep 40 + # Wait for job pods and specific log lines to appear (60s total) - MUST succeed or test fails + LOGS_READY=false + for i in {1..20}; do + # Get any pod from the job (jobs create pods that may be in Error state) + POD_NAME=$(kubectl get pods -n app-12 -l job-name=java-api-checker -o jsonpath='{.items[0].metadata.name}' 2>/dev/null) + if [ -n "$POD_NAME" ] && kubectl logs "$POD_NAME" -n app-12 2>/dev/null | grep -q "Target host: prod-db, port: 3333" && kubectl logs "$POD_NAME" -n app-12 2>/dev/null | grep -q "FATAL: Unable to connect to required database"; then + echo "✅ Required log lines detected in job pod!" + LOGS_READY=true + break + else + echo "⏳ Attempt $i/20: waiting for job pod with specific log lines, checking in 3s..." + sleep 3 + fi + done + if [ "$LOGS_READY" = false ]; then + echo "❌ Required log lines not found after 60s" + kubectl get pods -n app-12 + POD_NAME=$(kubectl get pods -n app-12 -l job-name=java-api-checker -o jsonpath='{.items[0].metadata.name}' 2>/dev/null) + if [ -n "$POD_NAME" ]; then + kubectl logs "$POD_NAME" -n app-12 --tail=50 + fi + exit 1 + fi after_test: | kubectl delete -f ./job.yaml kubectl delete secret java-api-checker-logs-script -n app-12 --ignore-not-found diff --git a/tests/llm/fixtures/test_ask_holmes/73a_time_window_anomaly/generate_logs.py b/tests/llm/fixtures/test_ask_holmes/73a_time_window_anomaly/generate_logs.py index da2432f96b..359c4b00dc 100644 --- a/tests/llm/fixtures/test_ask_holmes/73a_time_window_anomaly/generate_logs.py +++ b/tests/llm/fixtures/test_ask_holmes/73a_time_window_anomaly/generate_logs.py @@ -93,6 +93,17 @@ def main(): if random.random() < 0.1: current_time += timedelta(seconds=random.randint(1, 5)) + # add a unique log message in the end so we'll know the pod is ready. + current_time += timedelta(seconds=random.randint(1, 5)) + print( + generate_log_entry( + current_time, + status="success", + message="Job executed successfully in 167ms.", + ), + flush=True, + ) + # Keep pod running while True: time.sleep(3600) diff --git a/tests/llm/fixtures/test_ask_holmes/73a_time_window_anomaly/manifest.yaml b/tests/llm/fixtures/test_ask_holmes/73a_time_window_anomaly/manifest.yaml index 7f4c268085..750f20a60f 100644 --- a/tests/llm/fixtures/test_ask_holmes/73a_time_window_anomaly/manifest.yaml +++ b/tests/llm/fixtures/test_ask_holmes/73a_time_window_anomaly/manifest.yaml @@ -2,7 +2,7 @@ apiVersion: apps/v1 kind: Deployment metadata: name: task-scheduler - namespace: namespace-73a + namespace: app-73a labels: app: task-scheduler spec: diff --git a/tests/llm/fixtures/test_ask_holmes/73a_time_window_anomaly/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/73a_time_window_anomaly/test_case.yaml index 541c52fcd5..c6c3e5077d 100644 --- a/tests/llm/fixtures/test_ask_holmes/73a_time_window_anomaly/test_case.yaml +++ b/tests/llm/fixtures/test_ask_holmes/73a_time_window_anomaly/test_case.yaml @@ -1,4 +1,4 @@ -user_prompt: "Were there issues with the task-scheduler pod in namespace-73a?" +user_prompt: "Were there issues with the task-scheduler pod in app-73a?" tags: - logs - context_window @@ -8,11 +8,38 @@ expected_output: - Task scheduler pod experienced temporary issues between when it could not connect to an external API possibly due to a maintenance window before_test: | # Create namespace first since the secret depends on it (|| true ignores if it already exists) - kubectl create namespace namespace-73a || true - kubectl create secret generic task-scheduler-logs-script --from-file=generate_logs.py=./generate_logs.py -n namespace-73a --dry-run=client -o yaml | kubectl apply -f - + kubectl create namespace app-73a || true + kubectl create secret generic task-scheduler-logs-script --from-file=generate_logs.py=./generate_logs.py -n app-73a --dry-run=client -o yaml | kubectl apply -f - kubectl apply -f ./manifest.yaml - sleep 50 + # Wait for pod and all required log lines to appear (60s total) - MUST succeed or test fails + LOGS_READY=false + for i in {1..20}; do + # First check if pod exists and is ready + if kubectl wait --for=condition=ready pod -l app=task-scheduler -n app-73a --timeout=1s 2>/dev/null; then + # Pod is ready, now get its name and check for all three required log lines + POD_NAME=$(kubectl get pods -n app-73a -l app=task-scheduler -o jsonpath='{.items[0].metadata.name}' 2>/dev/null) + if [ -n "$POD_NAME" ] && kubectl logs "$POD_NAME" -n app-73a 2>/dev/null | grep -q "Detected repeated failures during 03:00-03:05 window" && kubectl logs "$POD_NAME" -n app-73a 2>/dev/null | grep -q "System health check passed" && kubectl logs "$POD_NAME" -n app-73a 2>/dev/null | grep -q "Job executed successfully in 167ms\."; then + echo "✅ All required log lines detected!" + LOGS_READY=true + break + else + echo "⏳ Attempt $i/20: pod ready but not all log lines found yet, checking in 3s..." + fi + else + echo "⏳ Attempt $i/20: waiting for pod to be ready, checking in 3s..." + fi + sleep 3 + done + if [ "$LOGS_READY" = false ]; then + echo "❌ Required log lines not found after 60s" + POD_NAME=$(kubectl get pods -n app-73a -l app=task-scheduler -o jsonpath='{.items[0].metadata.name}' 2>/dev/null) + if [ -n "$POD_NAME" ]; then + kubectl logs "$POD_NAME" -n app-73a --tail=50 + fi + kubectl get pods -n app-73a + exit 1 + fi after_test: | kubectl delete -f ./manifest.yaml - kubectl delete secret task-scheduler-logs-script -n namespace-73a --ignore-not-found - kubectl delete namespace namespace-73a --ignore-not-found + kubectl delete secret task-scheduler-logs-script -n app-73a --ignore-not-found + kubectl delete namespace app-73a --ignore-not-found diff --git a/tests/llm/fixtures/test_ask_holmes/73b_time_window_anomaly/generate_logs.py b/tests/llm/fixtures/test_ask_holmes/73b_time_window_anomaly/generate_logs.py index da2432f96b..359c4b00dc 100644 --- a/tests/llm/fixtures/test_ask_holmes/73b_time_window_anomaly/generate_logs.py +++ b/tests/llm/fixtures/test_ask_holmes/73b_time_window_anomaly/generate_logs.py @@ -93,6 +93,17 @@ def main(): if random.random() < 0.1: current_time += timedelta(seconds=random.randint(1, 5)) + # add a unique log message in the end so we'll know the pod is ready. + current_time += timedelta(seconds=random.randint(1, 5)) + print( + generate_log_entry( + current_time, + status="success", + message="Job executed successfully in 167ms.", + ), + flush=True, + ) + # Keep pod running while True: time.sleep(3600) diff --git a/tests/llm/fixtures/test_ask_holmes/73b_time_window_anomaly/manifest.yaml b/tests/llm/fixtures/test_ask_holmes/73b_time_window_anomaly/manifest.yaml index d7bef76b43..9a6122d621 100644 --- a/tests/llm/fixtures/test_ask_holmes/73b_time_window_anomaly/manifest.yaml +++ b/tests/llm/fixtures/test_ask_holmes/73b_time_window_anomaly/manifest.yaml @@ -2,7 +2,7 @@ apiVersion: apps/v1 kind: Deployment metadata: name: cron-scheduler - namespace: namespace-73b + namespace: app-73b labels: app: cron-scheduler spec: diff --git a/tests/llm/fixtures/test_ask_holmes/73b_time_window_anomaly/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/73b_time_window_anomaly/test_case.yaml index 17a459c4c7..0751aa3623 100644 --- a/tests/llm/fixtures/test_ask_holmes/73b_time_window_anomaly/test_case.yaml +++ b/tests/llm/fixtures/test_ask_holmes/73b_time_window_anomaly/test_case.yaml @@ -1,4 +1,4 @@ -user_prompt: "Were there issues with the cron-scheduler pod in namespace-73b?" +user_prompt: "Were there issues with the cron-scheduler pod in app-73b?" tags: - logs - context_window @@ -9,11 +9,38 @@ expected_output: - Answer MUST include the time window of the issue (03:00-03:05) before_test: | # Create namespace first since the secret depends on it (|| true ignores if it already exists) - kubectl create namespace namespace-73b || true - kubectl create secret generic cron-scheduler-logs-script --from-file=generate_logs.py=./generate_logs.py -n namespace-73b --dry-run=client -o yaml | kubectl apply -f - + kubectl create namespace app-73b || true + kubectl create secret generic cron-scheduler-logs-script --from-file=generate_logs.py=./generate_logs.py -n app-73b --dry-run=client -o yaml | kubectl apply -f - kubectl apply -f ./manifest.yaml - sleep 50 + # Wait for pod and all required log lines to appear (60s total) - MUST succeed or test fails + LOGS_READY=false + for i in {1..20}; do + # First check if pod exists and is ready + if kubectl wait --for=condition=ready pod -l app=cron-scheduler -n app-73b --timeout=1s 2>/dev/null; then + # Pod is ready, now get its name and check for all three required log lines + POD_NAME=$(kubectl get pods -n app-73b -l app=cron-scheduler -o jsonpath='{.items[0].metadata.name}' 2>/dev/null) + if [ -n "$POD_NAME" ] && kubectl logs "$POD_NAME" -n app-73b 2>/dev/null | grep -q "Detected repeated failures during 03:00-03:05 window" && kubectl logs "$POD_NAME" -n app-73b 2>/dev/null | grep -q "System health check passed" && kubectl logs "$POD_NAME" -n app-73b 2>/dev/null | grep -q "Job executed successfully in 167ms\."; then + echo "✅ All required log lines detected!" + LOGS_READY=true + break + else + echo "⏳ Attempt $i/20: pod ready but not all log lines found yet, checking in 3s..." + fi + else + echo "⏳ Attempt $i/20: waiting for pod to be ready, checking in 3s..." + fi + sleep 3 + done + if [ "$LOGS_READY" = false ]; then + echo "❌ Required log lines not found after 60s" + POD_NAME=$(kubectl get pods -n app-73b -l app=cron-scheduler -o jsonpath='{.items[0].metadata.name}' 2>/dev/null) + if [ -n "$POD_NAME" ]; then + kubectl logs "$POD_NAME" -n app-73b --tail=50 + fi + kubectl get pods -n app-73b + exit 1 + fi after_test: | kubectl delete -f ./manifest.yaml - kubectl delete secret cron-scheduler-logs-script -n namespace-73b --ignore-not-found - kubectl delete namespace namespace-73b --ignore-not-found + kubectl delete secret cron-scheduler-logs-script -n app-73b --ignore-not-found + kubectl delete namespace app-73b --ignore-not-found From eff66a16639d880de74c5faa2527b1fcecaef110 Mon Sep 17 00:00:00 2001 From: Natan Yellin Date: Wed, 24 Dec 2025 12:37:01 +0200 Subject: [PATCH 13/27] Add retention warning to dev image PR comments (#1224) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ## Summary - add warning about 30-day retention for temporary dev Docker images - include commands to copy temporary images into permanent registries ## Testing - not run (not needed) ------ [Codex Task](https://chatgpt.com/codex/tasks/task_b_694a33345d548327b15a2875460fff3f) ## Summary by CodeRabbit * **Chores** * Updated CI/CD workflow messages to use shorter image tags with a consistent marker prefix. * Enhanced PR/commit comment handling to update existing comments when the marker exists, avoiding duplicates. * Expanded message content with guidance that temporary images are deleted after 30 days and how to push to a permanent registry. * Added Helm patch instructions for two charts to set image tags and registries. ✏️ Tip: You can customize this high-level summary in your review settings. --------- Signed-off-by: Codex Signed-off-by: Filip Grebowski --- .github/workflows/docker-dev-images.yaml | 79 ++++++++++++++++++++---- 1 file changed, 68 insertions(+), 11 deletions(-) diff --git a/.github/workflows/docker-dev-images.yaml b/.github/workflows/docker-dev-images.yaml index 333bafcf94..2e1e85286f 100644 --- a/.github/workflows/docker-dev-images.yaml +++ b/.github/workflows/docker-dev-images.yaml @@ -65,17 +65,39 @@ jobs: const sha = context.sha; const shortSha = `${{ steps.short_sha.outputs.sha_short }}`; - const fullTag = `us-central1-docker.pkg.dev/robusta-development/temporary-builds/holmes:${sha}`; const shortTag = `us-central1-docker.pkg.dev/robusta-development/temporary-builds/holmes:${shortSha}`; const registryLink = 'https://console.cloud.google.com/artifacts/docker/robusta-development/us-central1/temporary-builds/holmes?project=robusta-development'; + const marker = 'Dev Docker images are ready for this commit:'; const message = [ - 'Dev Docker images are ready for this commit:', + marker, '', - `- [${fullTag}](${registryLink})`, `- [${shortTag}](${registryLink})`, '', - 'Use either tag to pull the image for testing.' + 'Use this tag to pull the image for testing.', + '', + '⚠️ Temporary images are deleted after 30 days. Copy to a permanent registry before using them:', + '```bash', + 'gcloud auth configure-docker us-central1-docker.pkg.dev', + `docker pull ${shortTag}`, + `docker tag ${shortTag} me-west1-docker.pkg.dev/robusta-development/development/holmes-dev:${shortSha}`, + `docker push me-west1-docker.pkg.dev/robusta-development/development/holmes-dev:${shortSha}`, + '```', + '', + 'Patch Helm values in one line (choose the chart you use):', + '- HolmesGPT chart:', + '```bash', + 'helm upgrade --install holmesgpt ./helm/holmes \\', + ' --set registry=me-west1-docker.pkg.dev/robusta-development/development \\', + ` --set image=holmes-dev:${shortSha}`, + '```', + '- Robusta wrapper chart:', + '```bash', + 'helm upgrade --install robusta robusta/robusta \\', + ' --reuse-values \\', + ' --set holmes.registry=me-west1-docker.pkg.dev/robusta-development/development \\', + ` --set holmes.image=holmes-dev:${shortSha}`, + '```', ].join('\n'); const prs = await github.paginate(github.rest.pulls.list, { @@ -86,19 +108,54 @@ jobs: }); if (prs.length > 0) { - await github.rest.issues.createComment({ + const issue_number = prs[0].number; + const existingComments = await github.paginate(github.rest.issues.listComments, { owner, repo, - issue_number: prs[0].number, - body: message, + issue_number, }); - core.info(`Commented on PR #${prs[0].number}`); + const existing = existingComments.find(c => c.body?.includes(marker)); + + if (existing) { + await github.rest.issues.updateComment({ + owner, + repo, + comment_id: existing.id, + body: message, + }); + core.info(`Updated existing PR comment #${existing.id}`); + } else { + await github.rest.issues.createComment({ + owner, + repo, + issue_number, + body: message, + }); + core.info(`Commented on PR #${issue_number}`); + } } else { - await github.rest.repos.createCommitComment({ + const commitComments = await github.paginate(github.rest.repos.listCommentsForCommit, { owner, repo, commit_sha: sha, - body: message, }); - core.info('Commented on commit'); + const existing = commitComments.find(c => c.body?.includes(marker)); + + if (existing) { + await github.rest.repos.updateCommitComment({ + owner, + repo, + comment_id: existing.id, + body: message, + }); + core.info(`Updated existing commit comment #${existing.id}`); + } else { + await github.rest.repos.createCommitComment({ + owner, + repo, + commit_sha: sha, + body: message, + }); + core.info('Commented on commit'); + } } From bdd40888bafb952ac6b75aa8302f604c5fdf645a Mon Sep 17 00:00:00 2001 From: Tomer Date: Wed, 24 Dec 2025 13:14:17 +0200 Subject: [PATCH 14/27] fixing failures due to non standard pod name in eval 101. test 100b is for testing nonstandard labels. (#1237) Signed-off-by: Tomer Keshet Signed-off-by: Filip Grebowski --- .../promtail-config.yaml | 4 ++-- .../101_loki_historical_logs_pod_deleted/toolsets.yaml | 5 +---- 2 files changed, 3 insertions(+), 6 deletions(-) diff --git a/tests/llm/fixtures/test_ask_holmes/101_loki_historical_logs_pod_deleted/promtail-config.yaml b/tests/llm/fixtures/test_ask_holmes/101_loki_historical_logs_pod_deleted/promtail-config.yaml index a0a10c0940..9da5563987 100644 --- a/tests/llm/fixtures/test_ask_holmes/101_loki_historical_logs_pod_deleted/promtail-config.yaml +++ b/tests/llm/fixtures/test_ask_holmes/101_loki_historical_logs_pod_deleted/promtail-config.yaml @@ -36,11 +36,11 @@ data: level: level message: message service: service - pod_name: pod + pod: pod - timestamp: source: timestamp format: RFC3339 - labels: level: service: - pod_name: + pod: diff --git a/tests/llm/fixtures/test_ask_holmes/101_loki_historical_logs_pod_deleted/toolsets.yaml b/tests/llm/fixtures/test_ask_holmes/101_loki_historical_logs_pod_deleted/toolsets.yaml index 43cb312e80..115bd19302 100644 --- a/tests/llm/fixtures/test_ask_holmes/101_loki_historical_logs_pod_deleted/toolsets.yaml +++ b/tests/llm/fixtures/test_ask_holmes/101_loki_historical_logs_pod_deleted/toolsets.yaml @@ -7,7 +7,4 @@ toolsets: enabled: true config: url: http://localhost:3101 - api_key: "" # No auth needed for local Loki - labels: - pod: pod_name # Match the label used by Promtail - namespace: namespace + api_key: "" From 5d9da0da4c36f503ab77bad82fbf961687daf8c2 Mon Sep 17 00:00:00 2001 From: Tomer Date: Wed, 24 Dec 2025 14:09:07 +0200 Subject: [PATCH 15/27] ad 176_network_policy_blocking_traffic_no_runbooks based on test 84 (#1236) Signed-off-by: Filip Grebowski --- .../backend.yaml | 67 +++++++++++ .../frontend.yaml | 39 +++++++ .../manifest.yaml | 107 ++++++++++++++++++ .../test_case.yaml | 37 ++++++ .../toolsets.yaml | 5 + 5 files changed, 255 insertions(+) create mode 100644 tests/llm/fixtures/test_ask_holmes/176_network_policy_blocking_traffic_no_runbooks/backend.yaml create mode 100644 tests/llm/fixtures/test_ask_holmes/176_network_policy_blocking_traffic_no_runbooks/frontend.yaml create mode 100644 tests/llm/fixtures/test_ask_holmes/176_network_policy_blocking_traffic_no_runbooks/manifest.yaml create mode 100644 tests/llm/fixtures/test_ask_holmes/176_network_policy_blocking_traffic_no_runbooks/test_case.yaml create mode 100644 tests/llm/fixtures/test_ask_holmes/176_network_policy_blocking_traffic_no_runbooks/toolsets.yaml diff --git a/tests/llm/fixtures/test_ask_holmes/176_network_policy_blocking_traffic_no_runbooks/backend.yaml b/tests/llm/fixtures/test_ask_holmes/176_network_policy_blocking_traffic_no_runbooks/backend.yaml new file mode 100644 index 0000000000..46ec545f2d --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/176_network_policy_blocking_traffic_no_runbooks/backend.yaml @@ -0,0 +1,67 @@ +apiVersion: v1 +kind: Namespace +metadata: + name: app-176 +--- +# Backend deployment +apiVersion: apps/v1 +kind: Deployment +metadata: + name: backend + namespace: app-176 +spec: + replicas: 1 + selector: + matchLabels: + app: backend + template: + metadata: + labels: + app: backend + tier: backend + spec: + containers: + - name: backend + image: nginx:alpine + ports: + - containerPort: 80 + resources: + requests: + memory: "64Mi" + cpu: "10m" + limits: + memory: "64Mi" +--- +# Backend service +apiVersion: v1 +kind: Service +metadata: + name: backend-service + namespace: app-176 +spec: + selector: + app: backend + ports: + - port: 80 + targetPort: 80 +--- +# Network Policy that blocks frontend->backend traffic +apiVersion: networking.k8s.io/v1 +kind: NetworkPolicy +metadata: + name: backend-network-policy + namespace: app-176 +spec: + podSelector: + matchLabels: + app: backend + policyTypes: + - Ingress + ingress: + - from: + - podSelector: + matchLabels: + tier: backend # Only allows traffic from pods with tier=backend + ports: + - protocol: TCP + port: 80 diff --git a/tests/llm/fixtures/test_ask_holmes/176_network_policy_blocking_traffic_no_runbooks/frontend.yaml b/tests/llm/fixtures/test_ask_holmes/176_network_policy_blocking_traffic_no_runbooks/frontend.yaml new file mode 100644 index 0000000000..e382fc5fd5 --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/176_network_policy_blocking_traffic_no_runbooks/frontend.yaml @@ -0,0 +1,39 @@ +# Frontend deployment +apiVersion: apps/v1 +kind: Deployment +metadata: + name: frontend + namespace: app-176 +spec: + replicas: 1 + selector: + matchLabels: + app: frontend + template: + metadata: + labels: + app: frontend + tier: frontend + spec: + containers: + - name: frontend + image: busybox + command: ["/bin/sh"] + args: + - -c + - | + while true; do + echo "Trying to connect to backend-service..." + if wget -O- http://backend-service:80 -T 5; then + echo "Success!" + else + echo "ERROR: Connection timeout to backend-service!" + fi + sleep 15 + done + resources: + requests: + memory: "64Mi" + cpu: "10m" + limits: + memory: "64Mi" diff --git a/tests/llm/fixtures/test_ask_holmes/176_network_policy_blocking_traffic_no_runbooks/manifest.yaml b/tests/llm/fixtures/test_ask_holmes/176_network_policy_blocking_traffic_no_runbooks/manifest.yaml new file mode 100644 index 0000000000..ca8f93bd3c --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/176_network_policy_blocking_traffic_no_runbooks/manifest.yaml @@ -0,0 +1,107 @@ +apiVersion: v1 +kind: Namespace +metadata: + name: app-176 +--- +# Backend deployment +apiVersion: apps/v1 +kind: Deployment +metadata: + name: backend + namespace: app-176 +spec: + replicas: 1 + selector: + matchLabels: + app: backend + template: + metadata: + labels: + app: backend + tier: backend + spec: + containers: + - name: backend + image: nginx:alpine + ports: + - containerPort: 80 + resources: + requests: + memory: "64Mi" + cpu: "10m" + limits: + memory: "64Mi" +--- +# Backend service +apiVersion: v1 +kind: Service +metadata: + name: backend-service + namespace: app-176 +spec: + selector: + app: backend + ports: + - port: 80 + targetPort: 80 +--- +# Frontend deployment +apiVersion: apps/v1 +kind: Deployment +metadata: + name: frontend + namespace: app-176 +spec: + replicas: 1 + selector: + matchLabels: + app: frontend + template: + metadata: + labels: + app: frontend + tier: frontend + spec: + containers: + - name: frontend + image: busybox + command: ["/bin/sh"] + args: + - -c + - | + while true; do + echo "Trying to connect to backend-service..." + if wget -O- http://backend-service:80 -T 5; then + echo "Success!" + else + echo "ERROR: Connection timeout to backend-service!" + fi + sleep 15 + done + resources: + requests: + memory: "64Mi" + cpu: "10m" + limits: + memory: "64Mi" +--- +# Network Policy that blocks frontend->backend traffic +apiVersion: networking.k8s.io/v1 +kind: NetworkPolicy +metadata: + name: backend-network-policy + namespace: app-176 +spec: + podSelector: + matchLabels: + app: backend + policyTypes: + - Ingress + ingress: + - from: + - podSelector: + matchLabels: + tier: backend # Only allows traffic from pods with tier=backend + ports: + - protocol: TCP + port: 80 diff --git a/tests/llm/fixtures/test_ask_holmes/176_network_policy_blocking_traffic_no_runbooks/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/176_network_policy_blocking_traffic_no_runbooks/test_case.yaml new file mode 100644 index 0000000000..d0a892314d --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/176_network_policy_blocking_traffic_no_runbooks/test_case.yaml @@ -0,0 +1,37 @@ +user_prompt: "Why is the frontend getting timeouts connecting to backend in namespace app-176?" +tags: + - kubernetes + - network + - hard +expected_output: + - Frontend getting timeouts because NetworkPolicy on backend only allows ingress from pods with tier backend label not frontend +before_test: | + # Apply backend resources first + kubectl apply -f ./backend.yaml + + # Wait for backend to be ready with retry loop + for i in {1..12}; do kubectl wait --for=condition=ready pod -l app=backend -n app-176 --timeout=5s 2>/dev/null && break || sleep 5; done + + # Apply frontend deployment after backend is ready + kubectl apply -f ./frontend.yaml + + # Wait for the frontend pod to report connection timeout error (60s total) - MUST succeed or test fails + TIMEOUT_ERROR_FOUND=false + for i in {1..30}; do + if kubectl logs -l app=frontend -n app-176 2>/dev/null | grep -q "ERROR: Connection timeout to backend-service!"; then + echo "✅ Connection timeout error detected!" + TIMEOUT_ERROR_FOUND=true + break + else + echo "⏳ Attempt $i/30: waiting for timeout error log, checking in 2s..." + sleep 2 + fi + done + if [ "$TIMEOUT_ERROR_FOUND" = false ]; then + echo "❌ Connection timeout error not found after 60s" + kubectl get pods -n app-176 + kubectl logs -l app=frontend -n app-176 --tail=20 + exit 1 + fi +after_test: | + kubectl delete namespace app-176 || true diff --git a/tests/llm/fixtures/test_ask_holmes/176_network_policy_blocking_traffic_no_runbooks/toolsets.yaml b/tests/llm/fixtures/test_ask_holmes/176_network_policy_blocking_traffic_no_runbooks/toolsets.yaml new file mode 100644 index 0000000000..0e8e39553c --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/176_network_policy_blocking_traffic_no_runbooks/toolsets.yaml @@ -0,0 +1,5 @@ +toolsets: + runbook: + enabled: false + internet: # Holmes tries to fetch networking/dns_troubleshooting_instructions.md via this because runbook is disabled + enabled: false From 98ca29212e62fa4d342091ab53b111f423804517 Mon Sep 17 00:00:00 2001 From: Tomer Date: Wed, 24 Dec 2025 14:19:59 +0200 Subject: [PATCH 16/27] faster reliable setup for some counting and Kubernetes tests (#1238) Signed-off-by: Tomer Keshet Signed-off-by: Filip Grebowski --- .../01_how_many_pods/test_case.yaml | 19 ++++++++++++- .../02_what_is_wrong_with_pod/test_case.yaml | 20 +++++++++++-- ...tatus.phase_Running_.metadata.name_pod.txt | 14 ---------- .../58_counting_pods_by_status/test_case.yaml | 28 +++++++++++++++++-- ...ata.labels.env_prod_.metadata.name_pod.txt | 14 ---------- ...ironment_production_.metadata.name_pod.txt | 7 ----- .../59_label_based_counting/test_case.yaml | 25 ++++++++++++++++- ...a.namespace_test-61_.metadata.name_pod.txt | 13 --------- .../61_exact_match_counting/test_case.yaml | 28 ++++++++++++++++++- 9 files changed, 112 insertions(+), 56 deletions(-) delete mode 100644 tests/llm/fixtures/test_ask_holmes/58_counting_pods_by_status/kubernetes_countitems_select_.metadata.namespace_test-58_and_.status.phase_Running_.metadata.name_pod.txt delete mode 100644 tests/llm/fixtures/test_ask_holmes/59_label_based_counting/kubernetes_countitems_select_.metadata.namespace_test-59_and_.metadata.labels.env_prod_.metadata.name_pod.txt delete mode 100644 tests/llm/fixtures/test_ask_holmes/59_label_based_counting/kubernetes_countitems_select_.metadata.namespace_test-59_and_.metadata.labels.environment_production_.metadata.name_pod.txt delete mode 100644 tests/llm/fixtures/test_ask_holmes/61_exact_match_counting/kubernetes_countitems_select_.metadata.namespace_test-61_.metadata.name_pod.txt diff --git a/tests/llm/fixtures/test_ask_holmes/01_how_many_pods/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/01_how_many_pods/test_case.yaml index 2070891b46..3c3ce7e942 100644 --- a/tests/llm/fixtures/test_ask_holmes/01_how_many_pods/test_case.yaml +++ b/tests/llm/fixtures/test_ask_holmes/01_how_many_pods/test_case.yaml @@ -3,7 +3,24 @@ expected_output: - There are 14 pods in the app-01 namespace before_test: | kubectl apply -f manifests.yaml - for i in {1..12}; do [ "$(kubectl get pods -l app=test-pod -n app-01 --no-headers 2>/dev/null | wc -l)" -eq 14 ] && kubectl wait --for=condition=ready pod -l app=test-pod -n app-01 --timeout=5s 2>/dev/null && break || sleep 5; done + # Wait for 14 pods to be created and ready (60s total) - MUST succeed or test fails + PODS_READY=false + for i in {1..12}; do + POD_COUNT=$(kubectl get pods -l app=test-pod -n app-01 --no-headers 2>/dev/null | wc -l) + if [ "$POD_COUNT" -eq 14 ] && kubectl wait --for=condition=ready pod -l app=test-pod -n app-01 --timeout=5s 2>/dev/null; then + echo "✅ All 14 pods created and ready!" + PODS_READY=true + break + else + echo "⏳ Attempt $i/12: $POD_COUNT/14 pods found, waiting 5s..." + sleep 5 + fi + done + if [ "$PODS_READY" = false ]; then + echo "❌ 14 ready pods not achieved after 60s" + kubectl get pods -n app-01 + exit 1 + fi after_test: | kubectl delete -f manifests.yaml tags: diff --git a/tests/llm/fixtures/test_ask_holmes/02_what_is_wrong_with_pod/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/02_what_is_wrong_with_pod/test_case.yaml index 95cb50b5d2..9c8e8f023d 100644 --- a/tests/llm/fixtures/test_ask_holmes/02_what_is_wrong_with_pod/test_case.yaml +++ b/tests/llm/fixtures/test_ask_holmes/02_what_is_wrong_with_pod/test_case.yaml @@ -12,8 +12,24 @@ tags: before_test: | kubectl create namespace app-02 kubectl apply -f ./manifest.yaml - # Wait for pod to exist and then for OOMKilled status - for i in {1..12}; do kubectl get pod giant-narwhal-6958c5bdd8-69gtn -n app-02 -o jsonpath='{.status.containerStatuses[0].state.terminated.reason}' 2>/dev/null | grep -q "OOMKilled" && break || sleep 5; done + # Wait for pod to be OOMKilled (60s total) - MUST succeed or test fails + OOM_KILLED=false + for i in {1..12}; do + if kubectl get pod giant-narwhal-6958c5bdd8-69gtn -n app-02 -o jsonpath='{.status.containerStatuses[0].state.terminated.reason}' 2>/dev/null | grep -q "OOMKilled"; then + echo "✅ Pod OOMKilled as expected!" + OOM_KILLED=true + break + else + echo "⏳ Attempt $i/12: waiting for pod to be OOMKilled, checking in 5s..." + sleep 5 + fi + done + if [ "$OOM_KILLED" = false ]; then + echo "❌ Pod not OOMKilled after 60s" + kubectl get pod giant-narwhal-6958c5bdd8-69gtn -n app-02 -o wide + kubectl describe pod giant-narwhal-6958c5bdd8-69gtn -n app-02 + exit 1 + fi after_test: | kubectl delete namespace app-02 diff --git a/tests/llm/fixtures/test_ask_holmes/58_counting_pods_by_status/kubernetes_countitems_select_.metadata.namespace_test-58_and_.status.phase_Running_.metadata.name_pod.txt b/tests/llm/fixtures/test_ask_holmes/58_counting_pods_by_status/kubernetes_countitems_select_.metadata.namespace_test-58_and_.status.phase_Running_.metadata.name_pod.txt deleted file mode 100644 index 4bc91dff84..0000000000 --- a/tests/llm/fixtures/test_ask_holmes/58_counting_pods_by_status/kubernetes_countitems_select_.metadata.namespace_test-58_and_.status.phase_Running_.metadata.name_pod.txt +++ /dev/null @@ -1,14 +0,0 @@ -{"toolset_name":"kubernetes/core","tool_name":"kubernetes_count","match_params":{"kind":"pod","jq_expr":".items[] | select(.metadata.namespace == \"test-58\" and .status.phase == \"Running\") | .metadata.name"}} -{"schema_version": "robusta:v1.0.0", "status": "success", "error": null, "return_code": 0, "data": null, "url": null, "invocation": "echo \"Command executed: kubectl get pod --all-namespaces -o json | jq -c -r '.items[] | select(.metadata.namespace == \"test-58\" and .status.phase == \"Running\") | .metadata.name'\"\necho \"---\"\n\n# Execute the command and capture both stdout and stderr separately\ntemp_error=$(mktemp)\nmatches=$(kubectl get pod --all-namespaces -o json 2>\"$temp_error\" | jq -c -r '.items[] | select(.metadata.namespace == \"test-58\" and .status.phase == \"Running\") | .metadata.name' 2>>\"$temp_error\")\nexit_code=$?\nerror_output=$(cat \"$temp_error\")\nrm -f \"$temp_error\"\n\nif [ $exit_code -ne 0 ]; then\n echo \"Error executing command (exit code: $exit_code):\"\n echo \"$error_output\"\n exit $exit_code\nelse\n # Show any stderr warnings even if command succeeded\n if [ -n \"$error_output\" ]; then\n echo \"Warnings/stderr output:\"\n echo \"$error_output\"\n echo \"---\"\n fi\n\n # Filter out empty lines for accurate count\n filtered_matches=$(echo \"$matches\" | grep -v '^$' | grep -v '^null$')\n if [ -z \"$filtered_matches\" ]; then\n count=0\n else\n count=$(echo \"$filtered_matches\" | wc -l)\n fi\n preview=$(echo \"$filtered_matches\" | head -n 10 | cut -c 1-200 | nl)\n\n echo \"$count results\"\n echo \"---\"\n echo \"A *preview* of results is shown below (up to 10 results, up to 200 chars):\"\n echo \"$preview\"\nfi", "params": {"kind": "pod", "jq_expr": ".items[] | select(.metadata.namespace == \"test-58\" and .status.phase == \"Running\") | .metadata.name"}} -Command executed: kubectl get pod --all-namespaces -o json | jq -c -r '.items[] | select(.metadata.namespace == test-58 and .status.phase == Running) | .metadata.name' ---- - 7 results ---- -A *preview* of results is shown below (up to 10 results, up to 200 chars): - 1 alpha-pod - 2 beta-pod - 3 delta-pod - 4 epsilon-pod - 5 gamma-pod - 6 iota-pod - 7 kappa-pod diff --git a/tests/llm/fixtures/test_ask_holmes/58_counting_pods_by_status/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/58_counting_pods_by_status/test_case.yaml index 8402a64b25..0d5f844b50 100644 --- a/tests/llm/fixtures/test_ask_holmes/58_counting_pods_by_status/test_case.yaml +++ b/tests/llm/fixtures/test_ask_holmes/58_counting_pods_by_status/test_case.yaml @@ -1,12 +1,34 @@ +# Note: This test might fail because CrashLoopBackOff status still counts as Running if the llm looks at status.phase user_prompt: "How many pods are actually running in the test-58 namespace?" expected_output: - 5 pods are running in the test-58 namespace before_test: | kubectl apply -f manifests.yaml - sleep 90 + # Wait for correct pod states (90s total) - MUST succeed or test fails + POD_STATES_READY=false + for i in {1..30}; do + # Count running pods (should be exactly 5) + RUNNING_COUNT=$(kubectl get pods -n test-58 --no-headers 2>/dev/null | grep " Running " | wc -l | tr -d ' ') + + # Count pods in error states (not Running, Pending, or ContainerCreating) + ERROR_COUNT=$(kubectl get pods -n test-58 --no-headers 2>/dev/null | grep -v "Running\|Pending\|ContainerCreating" | wc -l | tr -d ' ') + + # We expect 5 running pods and 5 pods in error states + if [ "$RUNNING_COUNT" = "5" ] && [ "$ERROR_COUNT" = "5" ]; then + echo "✅ Correct pod states detected: 5 running, 5 in error states!" + POD_STATES_READY=true + break + else + echo "⏳ Attempt $i/30: Running: $RUNNING_COUNT/5, Error: $ERROR_COUNT/5, checking in 3s..." + sleep 3 + fi + done + if [ "$POD_STATES_READY" = false ]; then + echo "❌ Expected pod states not reached after 90s" + kubectl get pods -n test-58 + exit 1 + fi after_test: | kubectl delete -f manifests.yaml -evaluation: - correctness: 0 tags: - counting diff --git a/tests/llm/fixtures/test_ask_holmes/59_label_based_counting/kubernetes_countitems_select_.metadata.namespace_test-59_and_.metadata.labels.env_prod_.metadata.name_pod.txt b/tests/llm/fixtures/test_ask_holmes/59_label_based_counting/kubernetes_countitems_select_.metadata.namespace_test-59_and_.metadata.labels.env_prod_.metadata.name_pod.txt deleted file mode 100644 index 838a17cb87..0000000000 --- a/tests/llm/fixtures/test_ask_holmes/59_label_based_counting/kubernetes_countitems_select_.metadata.namespace_test-59_and_.metadata.labels.env_prod_.metadata.name_pod.txt +++ /dev/null @@ -1,14 +0,0 @@ -{"toolset_name":"kubernetes/core","tool_name":"kubernetes_count","match_params":{"kind":"pod","jq_expr":".items[] | select(.metadata.namespace == \"test-59\" and .metadata.labels.env == \"prod\") | .metadata.name"}} -{"schema_version": "robusta:v1.0.0", "status": "success", "error": null, "return_code": 0, "data": null, "url": null, "invocation": "echo \"Command executed: kubectl get pod --all-namespaces -o json | jq -c -r '.items[] | select(.metadata.namespace == \"test-59\" and .metadata.labels.env == \"prod\") | .metadata.name'\"\necho \"---\"\n\n# Execute the command and capture both stdout and stderr separately\ntemp_error=$(mktemp)\nmatches=$(kubectl get pod --all-namespaces -o json 2>\"$temp_error\" | jq -c -r '.items[] | select(.metadata.namespace == \"test-59\" and .metadata.labels.env == \"prod\") | .metadata.name' 2>>\"$temp_error\")\nexit_code=$?\nerror_output=$(cat \"$temp_error\")\nrm -f \"$temp_error\"\n\nif [ $exit_code -ne 0 ]; then\n echo \"Error executing command (exit code: $exit_code):\"\n echo \"$error_output\"\n exit $exit_code\nelse\n # Show any stderr warnings even if command succeeded\n if [ -n \"$error_output\" ]; then\n echo \"Warnings/stderr output:\"\n echo \"$error_output\"\n echo \"---\"\n fi\n\n # Filter out empty lines for accurate count\n filtered_matches=$(echo \"$matches\" | grep -v '^$' | grep -v '^null$')\n if [ -z \"$filtered_matches\" ]; then\n count=0\n else\n count=$(echo \"$filtered_matches\" | wc -l)\n fi\n preview=$(echo \"$filtered_matches\" | head -n 10 | cut -c 1-200 | nl)\n\n echo \"$count results\"\n echo \"---\"\n echo \"A *preview* of results is shown below (up to 10 results, up to 200 chars):\"\n echo \"$preview\"\nfi", "params": {"kind": "pod", "jq_expr": ".items[] | select(.metadata.namespace == \"test-59\" and .metadata.labels.env == \"prod\") | .metadata.name"}} -Command executed: kubectl get pod --all-namespaces -o json | jq -c -r '.items[] | select(.metadata.namespace == test-59 and .metadata.labels.env == prod) | .metadata.name' ---- - 7 results ---- -A *preview* of results is shown below (up to 10 results, up to 200 chars): - 1 monitor-agent - 2 queue-processor - 3 service-a - 4 service-b - 5 service-c - 6 worker-x - 7 worker-y diff --git a/tests/llm/fixtures/test_ask_holmes/59_label_based_counting/kubernetes_countitems_select_.metadata.namespace_test-59_and_.metadata.labels.environment_production_.metadata.name_pod.txt b/tests/llm/fixtures/test_ask_holmes/59_label_based_counting/kubernetes_countitems_select_.metadata.namespace_test-59_and_.metadata.labels.environment_production_.metadata.name_pod.txt deleted file mode 100644 index bc780a2e4c..0000000000 --- a/tests/llm/fixtures/test_ask_holmes/59_label_based_counting/kubernetes_countitems_select_.metadata.namespace_test-59_and_.metadata.labels.environment_production_.metadata.name_pod.txt +++ /dev/null @@ -1,7 +0,0 @@ -{"toolset_name":"kubernetes/core","tool_name":"kubernetes_count","match_params":{"kind":"pod","jq_expr":".items[] | select(.metadata.namespace == \"test-59\" and .metadata.labels.environment == \"production\") | .metadata.name"}} -{"schema_version": "robusta:v1.0.0", "status": "success", "error": null, "return_code": 0, "data": null, "url": null, "invocation": "echo \"Command executed: kubectl get pod --all-namespaces -o json | jq -c -r '.items[] | select(.metadata.namespace == \"test-59\" and .metadata.labels.environment == \"production\") | .metadata.name'\"\necho \"---\"\n\n# Execute the command and capture both stdout and stderr separately\ntemp_error=$(mktemp)\nmatches=$(kubectl get pod --all-namespaces -o json 2>\"$temp_error\" | jq -c -r '.items[] | select(.metadata.namespace == \"test-59\" and .metadata.labels.environment == \"production\") | .metadata.name' 2>>\"$temp_error\")\nexit_code=$?\nerror_output=$(cat \"$temp_error\")\nrm -f \"$temp_error\"\n\nif [ $exit_code -ne 0 ]; then\n echo \"Error executing command (exit code: $exit_code):\"\n echo \"$error_output\"\n exit $exit_code\nelse\n # Show any stderr warnings even if command succeeded\n if [ -n \"$error_output\" ]; then\n echo \"Warnings/stderr output:\"\n echo \"$error_output\"\n echo \"---\"\n fi\n\n # Filter out empty lines for accurate count\n filtered_matches=$(echo \"$matches\" | grep -v '^$' | grep -v '^null$')\n if [ -z \"$filtered_matches\" ]; then\n count=0\n else\n count=$(echo \"$filtered_matches\" | wc -l)\n fi\n preview=$(echo \"$filtered_matches\" | head -n 10 | cut -c 1-200 | nl)\n\n echo \"$count results\"\n echo \"---\"\n echo \"A *preview* of results is shown below (up to 10 results, up to 200 chars):\"\n echo \"$preview\"\nfi", "params": {"kind": "pod", "jq_expr": ".items[] | select(.metadata.namespace == \"test-59\" and .metadata.labels.environment == \"production\") | .metadata.name"}} -Command executed: kubectl get pod --all-namespaces -o json | jq -c -r '.items[] | select(.metadata.namespace == test-59 and .metadata.labels.environment == production) | .metadata.name' ---- -0 results ---- -A *preview* of results is shown below (up to 10 results, up to 200 chars): diff --git a/tests/llm/fixtures/test_ask_holmes/59_label_based_counting/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/59_label_based_counting/test_case.yaml index c62d4745ef..e7fa388154 100644 --- a/tests/llm/fixtures/test_ask_holmes/59_label_based_counting/test_case.yaml +++ b/tests/llm/fixtures/test_ask_holmes/59_label_based_counting/test_case.yaml @@ -3,7 +3,30 @@ expected_output: - either 7, 7 pods, etc. before_test: | kubectl apply -f manifests.yaml - sleep 60 + # Wait for correct pod counts with labels (60s total) - MUST succeed or test fails + LABELED_PODS_READY=false + for i in {1..20}; do + # Count running pods with env=prod (should be 7) + PROD_COUNT=$(kubectl get pods -n test-59 -l env=prod --no-headers 2>/dev/null | grep " Running " | wc -l | tr -d ' ') + + # Count total running pods (should be 14) + TOTAL_COUNT=$(kubectl get pods -n test-59 --no-headers 2>/dev/null | grep " Running " | wc -l | tr -d ' ') + + # We expect 7 pods with env=prod and 14 total running pods + if [ "$PROD_COUNT" = "7" ] && [ "$TOTAL_COUNT" = "14" ]; then + echo "✅ Correct labeled pod counts detected: env=prod: 7/14 total!" + LABELED_PODS_READY=true + break + else + echo "⏳ Attempt $i/20: env=prod: $PROD_COUNT/7, total: $TOTAL_COUNT/14, checking in 3s..." + sleep 3 + fi + done + if [ "$LABELED_PODS_READY" = false ]; then + echo "❌ Expected labeled pod counts not reached after 60s" + kubectl get pods -n test-59 --show-labels + exit 1 + fi after_test: | kubectl delete -f manifests.yaml tags: diff --git a/tests/llm/fixtures/test_ask_holmes/61_exact_match_counting/kubernetes_countitems_select_.metadata.namespace_test-61_.metadata.name_pod.txt b/tests/llm/fixtures/test_ask_holmes/61_exact_match_counting/kubernetes_countitems_select_.metadata.namespace_test-61_.metadata.name_pod.txt deleted file mode 100644 index 9b449236c2..0000000000 --- a/tests/llm/fixtures/test_ask_holmes/61_exact_match_counting/kubernetes_countitems_select_.metadata.namespace_test-61_.metadata.name_pod.txt +++ /dev/null @@ -1,13 +0,0 @@ -{"toolset_name":"kubernetes/core","tool_name":"kubernetes_count","match_params":{"kind":"pod","jq_expr":".items[] | select(.metadata.namespace == \"test-61\") | .metadata.name"}} -{"schema_version": "robusta:v1.0.0", "status": "success", "error": null, "return_code": 0, "data": null, "url": null, "invocation": "echo \"Command executed: kubectl get pod --all-namespaces -o json | jq -c -r '.items[] | select(.metadata.namespace == \"test-61\") | .metadata.name'\"\necho \"---\"\n\n# Execute the command and capture both stdout and stderr separately\ntemp_error=$(mktemp)\nmatches=$(kubectl get pod --all-namespaces -o json 2>\"$temp_error\" | jq -c -r '.items[] | select(.metadata.namespace == \"test-61\") | .metadata.name' 2>>\"$temp_error\")\nexit_code=$?\nerror_output=$(cat \"$temp_error\")\nrm -f \"$temp_error\"\n\nif [ $exit_code -ne 0 ]; then\n echo \"Error executing command (exit code: $exit_code):\"\n echo \"$error_output\"\n exit $exit_code\nelse\n # Show any stderr warnings even if command succeeded\n if [ -n \"$error_output\" ]; then\n echo \"Warnings/stderr output:\"\n echo \"$error_output\"\n echo \"---\"\n fi\n\n # Filter out empty lines for accurate count\n filtered_matches=$(echo \"$matches\" | grep -v '^$' | grep -v '^null$')\n if [ -z \"$filtered_matches\" ]; then\n count=0\n else\n count=$(echo \"$filtered_matches\" | wc -l)\n fi\n preview=$(echo \"$filtered_matches\" | head -n 10 | cut -c 1-200 | nl)\n\n echo \"$count results\"\n echo \"---\"\n echo \"A *preview* of results is shown below (up to 10 results, up to 200 chars):\"\n echo \"$preview\"\nfi", "params": {"kind": "pod", "jq_expr": ".items[] | select(.metadata.namespace == \"test-61\") | .metadata.name"}} -Command executed: kubectl get pod --all-namespaces -o json | jq -c -r '.items[] | select(.metadata.namespace == test-61) | .metadata.name' ---- - 6 results ---- -A *preview* of results is shown below (up to 10 results, up to 200 chars): - 1 monitor-zeta - 2 queue-epsilon - 3 service-delta - 4 service-gamma - 5 worker-alpha - 6 worker-beta diff --git a/tests/llm/fixtures/test_ask_holmes/61_exact_match_counting/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/61_exact_match_counting/test_case.yaml index debbb07b69..b94538705c 100644 --- a/tests/llm/fixtures/test_ask_holmes/61_exact_match_counting/test_case.yaml +++ b/tests/llm/fixtures/test_ask_holmes/61_exact_match_counting/test_case.yaml @@ -4,7 +4,33 @@ expected_output: - either 6, 6 pods, etc. before_test: | kubectl apply -f manifests.yaml - sleep 60 + # Wait for correct pod counts in both namespaces (60s total) - MUST succeed or test fails + POD_COUNTS_READY=false + for i in {1..20}; do + # Count running pods in test-61 (should be 6) + TEST61_COUNT=$(kubectl get pods -n test-61 --no-headers 2>/dev/null | grep " Running " | wc -l | tr -d ' ') + + # Count running pods in test-611 (should be 9) + TEST611_COUNT=$(kubectl get pods -n test-611 --no-headers 2>/dev/null | grep " Running " | wc -l | tr -d ' ') + + # We expect 6 running pods in test-61 and 9 running pods in test-611 + if [ "$TEST61_COUNT" = "6" ] && [ "$TEST611_COUNT" = "9" ]; then + echo "✅ Correct pod counts detected: test-61: 6, test-611: 9!" + POD_COUNTS_READY=true + break + else + echo "⏳ Attempt $i/20: test-61: $TEST61_COUNT/6, test-611: $TEST611_COUNT/9, checking in 3s..." + sleep 3 + fi + done + if [ "$POD_COUNTS_READY" = false ]; then + echo "❌ Expected pod counts not reached after 60s" + echo "test-61 namespace:" + kubectl get pods -n test-61 + echo "test-611 namespace:" + kubectl get pods -n test-611 + exit 1 + fi after_test: | kubectl delete -f manifests.yaml tags: From d600384dbe87261f3acb6a3f5cea152ff35b4dfe Mon Sep 17 00:00:00 2001 From: Tomer Date: Wed, 24 Dec 2025 15:41:10 +0200 Subject: [PATCH 17/27] faster setup for 162_get_runbooks (#1239) Signed-off-by: Tomer Keshet Signed-off-by: Filip Grebowski --- .../app/payments-deployment.yaml | 4 +- .../162_get_runbooks/test_case.yaml | 40 ++++++++++++++++--- .../162_get_runbooks/toolsets.yaml | 31 -------------- 3 files changed, 36 insertions(+), 39 deletions(-) delete mode 100644 tests/llm/fixtures/test_ask_holmes/162_get_runbooks/toolsets.yaml diff --git a/tests/llm/fixtures/test_ask_holmes/162_get_runbooks/app/payments-deployment.yaml b/tests/llm/fixtures/test_ask_holmes/162_get_runbooks/app/payments-deployment.yaml index e0aa103a20..a4648e299d 100644 --- a/tests/llm/fixtures/test_ask_holmes/162_get_runbooks/app/payments-deployment.yaml +++ b/tests/llm/fixtures/test_ask_holmes/162_get_runbooks/app/payments-deployment.yaml @@ -49,13 +49,13 @@ spec: httpGet: path: /health port: 8080 - initialDelaySeconds: 30 + initialDelaySeconds: 2 periodSeconds: 10 readinessProbe: httpGet: path: /ready port: 8080 - initialDelaySeconds: 5 + initialDelaySeconds: 2 periodSeconds: 5 volumes: - name: app-code diff --git a/tests/llm/fixtures/test_ask_holmes/162_get_runbooks/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/162_get_runbooks/test_case.yaml index 0f826fe539..d521173694 100644 --- a/tests/llm/fixtures/test_ask_holmes/162_get_runbooks/test_case.yaml +++ b/tests/llm/fixtures/test_ask_holmes/162_get_runbooks/test_case.yaml @@ -15,12 +15,33 @@ before_test: | kubectl apply -f app/signup-deployment.yaml kubectl apply -f app/payments-deployment.yaml - # Wait for deployments to be ready - kubectl wait --for=condition=available --timeout=60s deployment/signup-service -n app-162 - kubectl wait --for=condition=available --timeout=60s deployment/payments-service -n app-162 - - # Give services time to start and fail - sleep 10 + # Wait for expected error logs to appear (60s total) - MUST succeed or test fails + LOGS_READY=false + for i in {1..20}; do + # Check signup-service for connection refused error + SIGNUP_ERROR=$(kubectl logs -l app=signup-service -n app-162 2>/dev/null | grep -c "Payments service check failed: ") + + # Check payments-service for stripe API key error + PAYMENT_ERROR=$(kubectl logs -l app=payments-service -n app-162 2>/dev/null | grep -c "Health check failed: Invalid Stripe API key") + + # We expect at least 1 occurrence of each error message + if [ "$SIGNUP_ERROR" -gt 0 ] && [ "$PAYMENT_ERROR" -gt 0 ]; then + echo "✅ Required error logs detected: signup errors: $SIGNUP_ERROR, payment errors: $PAYMENT_ERROR!" + LOGS_READY=true + break + else + echo "⏳ Attempt $i/20: signup errors: $SIGNUP_ERROR, payment errors: $PAYMENT_ERROR, checking in 3s..." + sleep 3 + fi + done + if [ "$LOGS_READY" = false ]; then + echo "❌ Required error logs not found after 60s" + echo "Signup-service logs:" + kubectl logs -l app=signup-service -n app-162 --tail=10 + echo "Payments-service logs:" + kubectl logs -l app=payments-service -n app-162 --tail=10 + exit 1 + fi after_test: | # Delete namespace @@ -29,3 +50,10 @@ after_test: | # verifies that the runbook is pulled from robusta and also global instructions are read and followed expected_output: | The runbook provides systematic debugging steps for signup service issues, emphasizing the need to check the payments service dependency first. The runbook should specifically mention checking Stripe API key validation in the payments service, as this is the root cause of the signup failures. Mentions contacting the interlock team for assistance. There is no mention of the capricorn team + +# Note: +# In the past we observed an instance in which the capricon team was mentioned. +# We did not try to fix it because we considered it a 'soft' failure. See below for what it looked like in the results. +# Contact: + # Stripe API issues: Contact interlock team + # Other issues: Contact capricorn team diff --git a/tests/llm/fixtures/test_ask_holmes/162_get_runbooks/toolsets.yaml b/tests/llm/fixtures/test_ask_holmes/162_get_runbooks/toolsets.yaml deleted file mode 100644 index 2433272750..0000000000 --- a/tests/llm/fixtures/test_ask_holmes/162_get_runbooks/toolsets.yaml +++ /dev/null @@ -1,31 +0,0 @@ -toolsets: - kubernetes/logs: - enabled: true - kubernetes/core: - enabled: true - helm/core: - enabled: false - internet: - enabled: false - kafka/admin: - enabled: false - aws/security: - enabled: false - aws/rds: - enabled: false - aks/core: - enabled: false - kubernetes/live-metrics: - enabled: false - kubernetes/kube-prometheus-stack: - enabled: false - kubernetes/kube-lineage-extras: - enabled: false - aks/node-health: - enabled: false - docker/core: - enabled: false - kubernetes/krew-extras: - enabled: false - runbook: - enabled: true From 1a73be595661a5217caa86f333a7cc86cd558699 Mon Sep 17 00:00:00 2001 From: Roi Glinik Date: Wed, 24 Dec 2025 16:47:33 +0200 Subject: [PATCH 18/27] ROB-2714 more default crd permissions (#1204) add default read permission to some kubernetes tools --------- Signed-off-by: Roi Glinik Signed-off-by: Filip Grebowski --- docs/data-sources/permissions.md | 58 ++++++++-- .../templates/holmesgpt-service-account.yaml | 106 ++++++++++++++---- helm/holmes/values.yaml | 11 ++ 3 files changed, 145 insertions(+), 30 deletions(-) diff --git a/docs/data-sources/permissions.md b/docs/data-sources/permissions.md index 420136a80a..72945e55a8 100644 --- a/docs/data-sources/permissions.md +++ b/docs/data-sources/permissions.md @@ -3,18 +3,54 @@ !!! note "In-Cluster Only" This page applies only to HolmesGPT running **inside** a Kubernetes cluster via Helm. For local CLI deployments, permissions are managed through your kubeconfig file. -HolmesGPT may require access to additional Kubernetes resources or CRDs for specific analyses. Permissions can be extended by modifying the ClusterRole rules. The default configuration has limited resource access. +HolmesGPT may require access to additional Kubernetes resources or CRDs for specific analyses. Permissions can be extended by modifying the ClusterRole rules. -## Common Scenarios for Adding Permissions +## Default CRD Permissions -1. **External Integrations and CRDs** - Access to custom resources from ArgoCD, Istio, etc. -2. **Additional Kubernetes resources** - Resources not included in the default permissions +HolmesGPT includes read-only permissions for common Kubernetes operators and tools by default. These can be individually enabled or disabled: + +=== "Holmes Helm Chart" -## Example Scenario: Adding Argo CD Permissions + ```yaml + crdPermissions: + argo: true + flux: true + kafka: true + keda: true + crossplane: true + istio: true + gatewayApi: true + velero: true + ``` + +=== "Robusta Helm Chart" + + ```yaml + enableHolmesGPT: true + holmes: + crdPermissions: + argo: true + flux: true + kafka: true + keda: true + crossplane: true + istio: true + gatewayApi: true + velero: true + ``` + +## Adding Custom Permissions + +For resources not covered by the default CRD permissions, you can add custom ClusterRole rules. + +### Common Scenarios + +1. **External Integrations and CRDs** - Access to custom resources from other operators +2. **Additional Kubernetes resources** - Resources not included in the default permissions -To enable HolmesGPT to analyze ArgoCD applications and projects, you need to add permissions for ArgoCD custom resources. +## Example: Adding Cert-Manager Permissions -### Steps to Add Permissions +To enable HolmesGPT to analyze cert-manager certificates and issuers (not included in default permissions), add custom ClusterRole rules: === "Holmes Helm Chart" @@ -22,8 +58,8 @@ To enable HolmesGPT to analyze ArgoCD applications and projects, you need to add ```yaml customClusterRoleRules: - - apiGroups: ["argoproj.io"] - resources: ["applications", "appprojects"] + - apiGroups: ["cert-manager.io"] + resources: ["certificates", "certificaterequests", "issuers", "clusterissuers"] verbs: ["get", "list", "watch"] ``` @@ -41,8 +77,8 @@ To enable HolmesGPT to analyze ArgoCD applications and projects, you need to add enableHolmesGPT: true holmes: customClusterRoleRules: - - apiGroups: ["argoproj.io"] - resources: ["applications", "appprojects"] + - apiGroups: ["cert-manager.io"] + resources: ["certificates", "certificaterequests", "issuers", "clusterissuers"] verbs: ["get", "list", "watch"] ``` diff --git a/helm/holmes/templates/holmesgpt-service-account.yaml b/helm/holmes/templates/holmesgpt-service-account.yaml index cc4dc5d796..e9cf813474 100644 --- a/helm/holmes/templates/holmesgpt-service-account.yaml +++ b/helm/holmes/templates/holmesgpt-service-account.yaml @@ -181,32 +181,100 @@ rules: - list - watch {{- end }} - # Prometheus CRDs + # Prometheus CRDs - apiGroups: - monitoring.coreos.com resources: - - alertmanagers - - alertmanagers/finalizers - - alertmanagers/status - - alertmanagerconfigs - - prometheuses - - prometheuses/finalizers - - prometheuses/status - - prometheusagents - - prometheusagents/finalizers - - prometheusagents/status - - thanosrulers - - thanosrulers/finalizers - - thanosrulers/status - - scrapeconfigs - - servicemonitors - - podmonitors - - probes - - prometheusrules + - "*" verbs: - get - list - watch +{{- if .Values.crdPermissions.argo }} + - apiGroups: + - argoproj.io + resources: + - "*" + verbs: + - get + - list + - watch +{{- end }} +{{- if .Values.crdPermissions.flux }} + - apiGroups: + - source.toolkit.fluxcd.io + - kustomize.toolkit.fluxcd.io + - helm.toolkit.fluxcd.io + - notification.toolkit.fluxcd.io + resources: + - "*" + verbs: + - get + - list + - watch +{{- end }} +{{- if .Values.crdPermissions.kafka }} + - apiGroups: + - kafka.strimzi.io + resources: + - "*" + verbs: + - get + - list + - watch +{{- end }} +{{- if .Values.crdPermissions.keda }} + - apiGroups: + - keda.sh + resources: + - "*" + verbs: + - get + - list + - watch +{{- end }} +{{- if .Values.crdPermissions.crossplane }} + - apiGroups: + - pkg.crossplane.io + - apiextensions.crossplane.io + resources: + - "*" + verbs: + - get + - list + - watch +{{- end }} +{{- if .Values.crdPermissions.istio }} + - apiGroups: + - networking.istio.io + - telemetry.istio.io + resources: + - "*" + verbs: + - get + - list + - watch +{{- end }} +{{- if .Values.crdPermissions.gatewayApi }} + - apiGroups: + - gateway.networking.k8s.io + resources: + - "*" + verbs: + - get + - list + - watch +{{- end }} +{{- if .Values.crdPermissions.velero }} + - apiGroups: + - velero.io + resources: + - "*" + verbs: + - get + - list + - watch +{{- end }} --- apiVersion: v1 diff --git a/helm/holmes/values.yaml b/helm/holmes/values.yaml index f125477877..4edc4a905b 100644 --- a/helm/holmes/values.yaml +++ b/helm/holmes/values.yaml @@ -30,6 +30,17 @@ customServiceAccountName: "" customClusterRoleRules: [] +# CRD permissions for common Kubernetes operators and tools +crdPermissions: + argo: true + flux: true + kafka: true + keda: true + crossplane: true + istio: true + gatewayApi: true + velero: true + enablePostProcessing: false postProcessingPrompt: "builtin://generic_post_processing.jinja2" openshift: false From ad973222409963ce99209a9e52f07b00311df5ed Mon Sep 17 00:00:00 2001 From: Ilia Lazebnik Date: Wed, 24 Dec 2025 17:33:52 -0500 Subject: [PATCH 19/27] helm: add oci release for charts (#1188) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Checklist for the toolset: - [ ] Toolset has unit tests where relevant - [ ] Toolset has both ask_holmes and investigate evals - [ ] Toolset has a documentation PR opened against the [robusta](https://github.com/robusta-dev/robusta) repo - [ ] Toolset has the correct is_default flag - [ ] Toolset returns a correct `get_example_config` - [ ] Toolset does a live health check in addition to checking for correct configuration - [ ] Create a demo video (if relevant) similar to https://github.com/robusta-dev/robusta/pull/1954 ## Summary by CodeRabbit * **Chores** * Enhanced deployment workflow with improved container registry integration and security enhancements for artifact distribution. ✏️ Tip: You can customize this high-level summary in your review settings. Signed-off-by: drfaust92 Co-authored-by: arik Signed-off-by: Filip Grebowski --- .github/workflows/build-docker-images.yaml | 17 +++++++++++++++-- 1 file changed, 15 insertions(+), 2 deletions(-) diff --git a/.github/workflows/build-docker-images.yaml b/.github/workflows/build-docker-images.yaml index f52acd6f5c..02ca00cb48 100644 --- a/.github/workflows/build-docker-images.yaml +++ b/.github/workflows/build-docker-images.yaml @@ -10,8 +10,9 @@ jobs: runs-on: ubuntu-latest permissions: - contents: 'read' - id-token: 'write' + contents: read + packages: write + id-token: write steps: - uses: actions/checkout@v4 @@ -77,3 +78,15 @@ jobs: - name: Upload helm chart run: | cd helm && ./upload_chart.sh + + - name: Login to GitHub Container Registry + uses: docker/login-action@v3 + with: + registry: ghcr.io + username: ${{ github.actor }} + password: ${{ secrets.GITHUB_TOKEN }} + - name: Push Helm chart to OCI registry + run: | + helm package helm/holmes + helm push holmes-${{github.ref_name}}.tgz oci://ghcr.io/${{ github.repository_owner }}/charts + From a0bca6561c9d6bda8f3e5c456faa0d19a425fe7d Mon Sep 17 00:00:00 2001 From: Natan Yellin Date: Thu, 25 Dec 2025 08:52:14 +0200 Subject: [PATCH 20/27] docs: use concrete Anthropic model example instead of placeholder (#1235) Make docs easier to copy-paste by replacing generic `` placeholder with actual example. Signed-off-by: Filip Grebowski --- docs/ai-providers/anthropic.md | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/docs/ai-providers/anthropic.md b/docs/ai-providers/anthropic.md index de5621379e..fbc310efd4 100644 --- a/docs/ai-providers/anthropic.md +++ b/docs/ai-providers/anthropic.md @@ -12,9 +12,11 @@ Get an [Anthropic API key](https://support.anthropic.com/en/articles/8114521-how ```bash export ANTHROPIC_API_KEY="your-anthropic-api-key" - holmes ask "what pods are failing?" --model="anthropic/" + holmes ask "what pods are failing?" --model="anthropic/claude-sonnet-4-5" ``` + **Note**: You can use any Anthropic model by changing the model name. See [Claude Models Overview](https://docs.claude.com/en/docs/about-claude/models/overview#latest-models-comparison){:target="_blank"} for available model names. + === "Holmes Helm Chart" **Create Kubernetes Secret:** @@ -99,9 +101,11 @@ Get an [Anthropic API key](https://support.anthropic.com/en/articles/8114521-how You can also pass the API key directly as a command-line parameter: ```bash -holmes ask "what pods are failing?" --model="anthropic/" --api-key="your-api-key" +holmes ask "what pods are failing?" --model="anthropic/claude-sonnet-4-5" --api-key="your-api-key" ``` +**Note**: You can use any Anthropic model by changing the model name. See [Claude Models Overview](https://docs.claude.com/en/docs/about-claude/models/overview#latest-models-comparison){:target="_blank"} for available model names. + ## Prompt Caching HolmesGPT adds Anthropic's prompt caching feature, which can significantly reduce costs and latency for repeated API calls with similar prompts. From 944c8d96e9e9cc424843d940e9a61befd0bf30c2 Mon Sep 17 00:00:00 2001 From: Tomer Date: Thu, 25 Dec 2025 13:29:45 +0200 Subject: [PATCH 21/27] add a regression test tag that is to be run in PRs instead of easy tests (#1241) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Signed-off-by: Tomer Keshet ## Summary by CodeRabbit * **New Features** * Added a "regression" test marker to identify critical regression tests. * **Chores** * CI test gating updated: consolidated pre-check determines whether regression tests run; environment and cluster setup now conditional. * Test cluster networking enhanced for Calico and increased pod capacity. * Test tooling forced to use bash for command execution. * Several tests retagged as regression; one test timeout increased. * **Documentation** * Updated marker guidance and parallel-run/infrastructure notes. ✏️ Tip: You can customize this high-level summary in your review settings. --------- Signed-off-by: Tomer Keshet Signed-off-by: Filip Grebowski --- .github/actions/setup-kind-cluster/action.yml | 163 ++++++++++++++++-- .github/workflows/eval-regression.yaml | 39 +++-- CLAUDE.md | 2 +- pyproject.toml | 8 +- .../09_crashpod/test_case.yaml | 1 + .../test_case.yaml | 3 +- .../12_job_crashing/test_case.yaml | 3 +- .../162_get_runbooks/test_case.yaml | 2 + .../test_case.yaml | 3 +- .../test_case.yaml | 1 + .../61_exact_match_counting/test_case.yaml | 1 + tests/llm/utils/commands.py | 1 + 12 files changed, 191 insertions(+), 36 deletions(-) diff --git a/.github/actions/setup-kind-cluster/action.yml b/.github/actions/setup-kind-cluster/action.yml index 9dbbf163c8..9cc1141225 100644 --- a/.github/actions/setup-kind-cluster/action.yml +++ b/.github/actions/setup-kind-cluster/action.yml @@ -17,7 +17,7 @@ runs: - name: Install KIND shell: bash run: | - curl -Lo ./kind https://kind.sigs.k8s.io/dl/v0.20.0/kind-linux-amd64 + curl -Lo ./kind https://kind.sigs.k8s.io/dl/v0.31.0/kind-linux-amd64 chmod +x ./kind sudo mv ./kind /usr/local/bin/kind kind version @@ -28,6 +28,8 @@ runs: cat < kind-config.yaml kind: Cluster apiVersion: kind.x-k8s.io/v1alpha4 + networking: + disableDefaultCNI: true # Disable default kindnet to install Calico nodes: - role: control-plane extraPortMappings: @@ -42,31 +44,154 @@ runs: kind: InitConfiguration nodeRegistration: kubeletExtraArgs: - max-pods: "200" + max-pods: "300" - | kind: KubeProxyConfiguration metricsBindAddress: "0.0.0.0:10249" - | kind: KubeletConfiguration - maxPods: 200 - - role: worker - kubeadmConfigPatches: - - | - kind: JoinConfiguration - nodeRegistration: - kubeletExtraArgs: - max-pods: "200" - - | - kind: KubeletConfiguration - maxPods: 200 + maxPods: 300 EOF - # Create cluster with resource limits appropriate for GitHub Actions (7GB RAM, 2 CPU) - kind create cluster --name ${{ inputs.cluster-name }} --config kind-config.yaml --wait 5m + # Create cluster without waiting for full readiness (CNI needed first) + kind create cluster --name ${{ inputs.cluster-name }} --config kind-config.yaml # Configure kubectl kubectl cluster-info --context kind-${{ inputs.cluster-name }} + - name: Install Calico CNI + shell: bash + run: | + set -e # Exit on any error + + # Calico version configuration + CALICO_VERSION="v3.31.3" + + echo "Installing Calico CNI ${CALICO_VERSION} for NetworkPolicy support..." + + # Install Calico operator + echo "Installing Calico operator..." + if ! kubectl create -f https://raw.githubusercontent.com/projectcalico/calico/${CALICO_VERSION}/manifests/tigera-operator.yaml; then + echo "❌ Failed to install Calico operator" + exit 1 + fi + + # Wait for operator to be ready + echo "Waiting for Calico operator to be ready..." + + # First, wait for operator pods to exist (max 60 attempts, 5s each = 300s total) + OPERATOR_PODS_FOUND=false + for i in $(seq 1 60); do + if kubectl get pods -n tigera-operator --no-headers 2>/dev/null | grep -q "tigera-operator"; then + echo "✅ Calico operator pods found!" + OPERATOR_PODS_FOUND=true + break + else + echo "⏳ Attempt $i/60: waiting for operator pods to be created, checking in 5s..." + sleep 5 + fi + done + + if [ "$OPERATOR_PODS_FOUND" = false ]; then + echo "❌ Calico operator pods failed to appear after 300 seconds" + kubectl get pods -n tigera-operator + kubectl describe pods -n tigera-operator + exit 1 + fi + + # Now wait for operator pods to be ready + echo "Waiting for operator pods to become ready..." + if ! kubectl wait --for=condition=Ready --timeout=300s -n tigera-operator pods --all; then + echo "❌ Calico operator failed to become ready" + kubectl get pods -n tigera-operator + kubectl describe pods -n tigera-operator + exit 1 + fi + + # Wait for operator to register CRDs + echo "Waiting for Calico operator CRDs to be registered..." + CRD_FOUND=false + for i in $(seq 1 60); do + if kubectl get crd installations.operator.tigera.io 2>/dev/null; then + echo "✅ Calico Installation CRD found!" + CRD_FOUND=true + break + else + echo "⏳ Attempt $i/60: waiting for Installation CRD to be registered, checking in 5s..." + sleep 5 + fi + done + + if [ "$CRD_FOUND" = false ]; then + echo "❌ Calico Installation CRD failed to be registered after 300 seconds" + kubectl get crds | grep -i tigera || echo "No tigera CRDs found" + kubectl get pods -n tigera-operator + kubectl describe pods -n tigera-operator + exit 1 + fi + + # Create custom resource for Calico installation optimized for KIND + echo "Creating Calico installation resource..." + if ! cat </dev/null | grep -q "calico"; then + echo "✅ Calico system pods found!" + CALICO_PODS_FOUND=true + break + else + echo "⏳ Attempt $i/60: waiting for Calico system pods to be created, checking in 5s..." + sleep 5 + fi + done + + if [ "$CALICO_PODS_FOUND" = false ]; then + echo "❌ Calico system pods failed to appear after 300 seconds" + echo "=== Calico system pod status ===" + kubectl get pods -n calico-system + echo "=== Calico installation status ===" + kubectl get installations.operator.tigera.io default -o yaml || echo "Installation resource not found" + exit 1 + fi + + # Now wait for Calico system pods to be ready + echo "Waiting for Calico system pods to become ready..." + if ! kubectl wait --for=condition=Ready --timeout=300s -n calico-system pods --all; then + echo "❌ Calico pods failed to become ready within 300 seconds" + echo "=== Calico system pod status ===" + kubectl get pods -n calico-system + echo "=== Calico installation status ===" + kubectl get installations.operator.tigera.io default -o yaml || echo "Installation resource not found" + echo "=== Calico system pod descriptions ===" + kubectl describe pods -n calico-system + exit 1 + fi + + + echo "✅ Calico CNI installed successfully with NetworkPolicy support" + - name: Wait for cluster to be ready if: inputs.wait-for-ready == 'true' shell: bash @@ -79,6 +204,8 @@ runs: # Wait for all system pods to be ready echo "Waiting for system pods to be ready..." kubectl wait --for=condition=Ready pods --all -n kube-system --timeout=300s + + # Verify cluster is working by creating a test pod echo "Creating test pod to verify cluster..." @@ -93,4 +220,8 @@ runs: kubectl get nodes -o wide echo "=== System pods ===" kubectl get pods -n kube-system - echo "=== Cluster is ready for tests ===" + echo "=== Calico pods ===" + kubectl get pods -n calico-system + echo "=== CNI configuration ===" + kubectl get installations.operator.tigera.io default -o yaml | grep -A 10 "status:" || echo "Installation status not available yet" + echo "=== Cluster is ready for tests with NetworkPolicy support ===" diff --git a/.github/workflows/eval-regression.yaml b/.github/workflows/eval-regression.yaml index 53b84df5a5..ade7346387 100644 --- a/.github/workflows/eval-regression.yaml +++ b/.github/workflows/eval-regression.yaml @@ -17,30 +17,37 @@ jobs: steps: - uses: actions/checkout@v4 - - name: Setup HolmesGPT environment - uses: ./.github/actions/setup-holmes-env - with: - python-version: '3.12' - install-kubectl: 'true' - - - name: Setup KIND cluster - uses: ./.github/actions/setup-kind-cluster - with: - cluster-name: 'kind' - wait-for-ready: 'true' - - name: Check if tests should run id: check-tests shell: bash run: | - if [[ "${{ github.event_name }}" != "pull_request" ]] || [[ -n "${{ secrets.AZURE_API_KEY }}" && -n "${{ secrets.BRAINTRUST_API_KEY }}" ]]; then + # Check if we have the required API keys to run LLM tests + # For PRs from forks, secrets are not available for security reasons + # For pushes to master, we always run regardless of secrets availability + # Note: We must check this in a step rather than job-level 'if' because + # the secrets context is not available in job-level conditions per GitHub Actions docs + if [[ "${{ github.event_name }}" != "pull_request" ]] || [[ -n "${{ secrets.BRAINTRUST_API_KEY }}" && (-n "${{ secrets.AZURE_API_KEY }}" || -n "${{ secrets.AWS_BEARER_TOKEN_BEDROCK }}") ]]; then echo "should-run=true" >> $GITHUB_OUTPUT - echo "✅ Running tests - either not a PR or all required secrets are available" + echo "✅ Running tests - either not a PR or required secrets are available" else echo "should-run=false" >> $GITHUB_OUTPUT echo "⏭️ Skipping tests - PR from fork without required secrets" fi + - name: Setup KIND cluster + if: steps.check-tests.outputs.should-run == 'true' + uses: ./.github/actions/setup-kind-cluster + with: + cluster-name: 'kind' + wait-for-ready: 'true' + + - name: Setup HolmesGPT environment + if: steps.check-tests.outputs.should-run == 'true' + uses: ./.github/actions/setup-holmes-env + with: + python-version: '3.12' + install-kubectl: 'true' + - name: Run tests if: steps.check-tests.outputs.should-run == 'true' id: evals @@ -61,7 +68,9 @@ jobs: EXPERIMENT_ID: github-${{ github.run_id }}.${{ github.run_number }}.${{ github.run_attempt }} GENERATE_REGRESSIONS_FILE: "true" run: | - poetry run pytest --no-cov tests/llm/test_ask_holmes.py tests/llm/test_investigate.py -s -n10 -m 'llm and easy and not no-cicd' + # Run regression tests - these are critical tests that must always pass + # Tests are marked with the 'regression' tag in their test_case.yaml files + poetry run pytest --no-cov tests/llm/test_ask_holmes.py tests/llm/test_investigate.py -s -n10 -m 'llm and regression' - name: Post evaluation results if: always() uses: ./.github/actions/post-eval-comment diff --git a/CLAUDE.md b/CLAUDE.md index dfb3b5e851..bbfa62e45f 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -197,7 +197,7 @@ RUN_LIVE=true poetry run pytest -m "llm and not easy" --no-cov # Non-regression **Available Test Markers (same as eval tags)**: Check in pyproject.toml and NEVER use a marker/tag that doesn't exist there. Ask the user before adding a new one. -**Important**: The `easy` marker identifies regression tests - these are the most important tests that should always pass. Run with `RUN_LIVE=true ITERATIONS=10 poetry run pytest -m "llm and easy"` to ensure stability. +**Important**: The `regression` marker identifies critical tests that must always pass in CI/CD. The `easy` marker is a legacy marker that contains broader regression tests. **Test Infrastructure Notes**: - All test state tracking uses pytest's `user_properties` to ensure compatibility with pytest-xdist parallel execution diff --git a/pyproject.toml b/pyproject.toml index 24baa680e4..8709641f5a 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -128,7 +128,13 @@ markers = [ "embeds: Tests ensuring embeds are as expected", "one-test: Runs only one simple eval", "loki: Loki toolset", - "coralogix: Runs coralogix evals" + "coralogix: Runs coralogix evals", + # REGRESSION TEST SELECTION CRITERIA: + # - Must pass 30+ iterations reliably with Sonnet-4.5 model + # - No external API dependencies (Loki OK, NewRelic/DataDog excluded) + # - Tests core HolmesGPT functionality that should never regress + # - Fast execution to avoid slowing CI/CD pipelines + "regression: Critical regression tests that must always pass" ] addopts = [ diff --git a/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml index 1fc128f2a7..249ac71f0c 100644 --- a/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml +++ b/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml @@ -52,3 +52,4 @@ tags: - easy - kubernetes - one-test + - regression diff --git a/tests/llm/fixtures/test_ask_holmes/101_loki_historical_logs_pod_deleted/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/101_loki_historical_logs_pod_deleted/test_case.yaml index e80658dbee..d9ff2b0da1 100644 --- a/tests/llm/fixtures/test_ask_holmes/101_loki_historical_logs_pod_deleted/test_case.yaml +++ b/tests/llm/fixtures/test_ask_holmes/101_loki_historical_logs_pod_deleted/test_case.yaml @@ -14,8 +14,9 @@ expected_output: tags: - logs - kubernetes - - hard + - easy - loki + - regression setup_timeout: 360 # Increased from 300 to handle pod deletion and Loki ingestion variability diff --git a/tests/llm/fixtures/test_ask_holmes/12_job_crashing/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/12_job_crashing/test_case.yaml index 26cf52be66..fe7dd1d7ab 100644 --- a/tests/llm/fixtures/test_ask_holmes/12_job_crashing/test_case.yaml +++ b/tests/llm/fixtures/test_ask_holmes/12_job_crashing/test_case.yaml @@ -36,4 +36,5 @@ after_test: | kubectl delete secret java-api-checker-logs-script -n app-12 --ignore-not-found kubectl delete namespace app-12 --ignore-not-found tags: - - hard + - easy + - regression diff --git a/tests/llm/fixtures/test_ask_holmes/162_get_runbooks/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/162_get_runbooks/test_case.yaml index d521173694..13a4468d4b 100644 --- a/tests/llm/fixtures/test_ask_holmes/162_get_runbooks/test_case.yaml +++ b/tests/llm/fixtures/test_ask_holmes/162_get_runbooks/test_case.yaml @@ -2,6 +2,8 @@ user_prompt: "My signup service is failing and users can't register. What should tags: - kubernetes - runbooks + - regression + - medium # simulates loading runbooks and global instructions from a cluster environment test_type: cluster diff --git a/tests/llm/fixtures/test_ask_holmes/176_network_policy_blocking_traffic_no_runbooks/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/176_network_policy_blocking_traffic_no_runbooks/test_case.yaml index d0a892314d..3f559b2dd2 100644 --- a/tests/llm/fixtures/test_ask_holmes/176_network_policy_blocking_traffic_no_runbooks/test_case.yaml +++ b/tests/llm/fixtures/test_ask_holmes/176_network_policy_blocking_traffic_no_runbooks/test_case.yaml @@ -2,7 +2,8 @@ user_prompt: "Why is the frontend getting timeouts connecting to backend in name tags: - kubernetes - network - - hard + - easy + - regression expected_output: - Frontend getting timeouts because NetworkPolicy on backend only allows ingress from pods with tier backend label not frontend before_test: | diff --git a/tests/llm/fixtures/test_ask_holmes/43_current_datetime_from_prompt/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/43_current_datetime_from_prompt/test_case.yaml index 7b7763c0ea..3bdde68026 100644 --- a/tests/llm/fixtures/test_ask_holmes/43_current_datetime_from_prompt/test_case.yaml +++ b/tests/llm/fixtures/test_ask_holmes/43_current_datetime_from_prompt/test_case.yaml @@ -7,3 +7,4 @@ mocked_date: "2025-06-23T11:34:00Z" tags: - datetime - easy + - regression diff --git a/tests/llm/fixtures/test_ask_holmes/61_exact_match_counting/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/61_exact_match_counting/test_case.yaml index b94538705c..c349055f4d 100644 --- a/tests/llm/fixtures/test_ask_holmes/61_exact_match_counting/test_case.yaml +++ b/tests/llm/fixtures/test_ask_holmes/61_exact_match_counting/test_case.yaml @@ -36,3 +36,4 @@ after_test: | tags: - counting - easy + - regression diff --git a/tests/llm/utils/commands.py b/tests/llm/utils/commands.py index b67748d3c0..39dd5bab79 100644 --- a/tests/llm/utils/commands.py +++ b/tests/llm/utils/commands.py @@ -156,6 +156,7 @@ def _invoke_command( result = subprocess.run( command, shell=True, + executable="/bin/bash", # Force bash instead of default /bin/sh capture_output=True, text=True, check=True, From 5827dcc74f020eb8745e9f153446d3678882ee6e Mon Sep 17 00:00:00 2001 From: Roi Glinik Date: Thu, 25 Dec 2025 18:53:33 +0200 Subject: [PATCH 22/27] ROB-2714 default crd (#1243) Signed-off-by: Roi Glinik Co-authored-by: arik Signed-off-by: Filip Grebowski --- docs/data-sources/permissions.md | 2 + .../templates/holmesgpt-service-account.yaml | 137 ++++++++++++++++-- helm/holmes/values.yaml | 3 +- 3 files changed, 132 insertions(+), 10 deletions(-) diff --git a/docs/data-sources/permissions.md b/docs/data-sources/permissions.md index 72945e55a8..28db861e89 100644 --- a/docs/data-sources/permissions.md +++ b/docs/data-sources/permissions.md @@ -21,6 +21,7 @@ HolmesGPT includes read-only permissions for common Kubernetes operators and too istio: true gatewayApi: true velero: true + externalSecrets: true ``` === "Robusta Helm Chart" @@ -37,6 +38,7 @@ HolmesGPT includes read-only permissions for common Kubernetes operators and too istio: true gatewayApi: true velero: true + externalSecrets: true ``` ## Adding Custom Permissions diff --git a/helm/holmes/templates/holmesgpt-service-account.yaml b/helm/holmes/templates/holmesgpt-service-account.yaml index e9cf813474..c0ab7b743c 100644 --- a/helm/holmes/templates/holmesgpt-service-account.yaml +++ b/helm/holmes/templates/holmesgpt-service-account.yaml @@ -185,7 +185,24 @@ rules: - apiGroups: - monitoring.coreos.com resources: - - "*" + - alertmanagers + - alertmanagers/finalizers + - alertmanagers/status + - alertmanagerconfigs + - prometheuses + - prometheuses/finalizers + - prometheuses/status + - prometheusagents + - prometheusagents/finalizers + - prometheusagents/status + - thanosrulers + - thanosrulers/finalizers + - thanosrulers/status + - scrapeconfigs + - servicemonitors + - podmonitors + - probes + - prometheusrules verbs: - get - list @@ -194,7 +211,18 @@ rules: - apiGroups: - argoproj.io resources: - - "*" + - applications + - applicationsets + - appprojects + - workflows + - workflowtemplates + - cronworkflows + - rollouts + - analysisruns + - analysistemplates + - experiments + - eventsources + - sensors verbs: - get - list @@ -203,11 +231,38 @@ rules: {{- if .Values.crdPermissions.flux }} - apiGroups: - source.toolkit.fluxcd.io + resources: + - gitrepositories + - helmrepositories + - helmcharts + - buckets + - ocirepositories + verbs: + - get + - list + - watch + - apiGroups: - kustomize.toolkit.fluxcd.io + resources: + - kustomizations + verbs: + - get + - list + - watch + - apiGroups: - helm.toolkit.fluxcd.io + resources: + - helmreleases + verbs: + - get + - list + - watch + - apiGroups: - notification.toolkit.fluxcd.io resources: - - "*" + - alerts + - providers + - receivers verbs: - get - list @@ -217,7 +272,14 @@ rules: - apiGroups: - kafka.strimzi.io resources: - - "*" + - kafkas + - kafkatopics + - kafkausers + - kafkaconnects + - kafkaconnectors + - kafkamirrormakers + - kafkabridges + - kafkarebalances verbs: - get - list @@ -227,7 +289,10 @@ rules: - apiGroups: - keda.sh resources: - - "*" + - scaledobjects + - scaledjobs + - triggerauthentications + - clustertriggerauthentications verbs: - get - list @@ -236,9 +301,19 @@ rules: {{- if .Values.crdPermissions.crossplane }} - apiGroups: - pkg.crossplane.io + resources: + - providers + - configurations + - functions + verbs: + - get + - list + - watch + - apiGroups: - apiextensions.crossplane.io resources: - - "*" + - compositions + - compositeresourcedefinitions verbs: - get - list @@ -247,9 +322,24 @@ rules: {{- if .Values.crdPermissions.istio }} - apiGroups: - networking.istio.io + resources: + - virtualservices + - destinationrules + - gateways + - serviceentries + - sidecars + - workloadentries + - workloadgroups + - proxyconfigs + - envoyfilters + verbs: + - get + - list + - watch + - apiGroups: - telemetry.istio.io resources: - - "*" + - telemetries verbs: - get - list @@ -259,7 +349,14 @@ rules: - apiGroups: - gateway.networking.k8s.io resources: - - "*" + - gatewayclasses + - gateways + - httproutes + - tcproutes + - tlsroutes + - udproutes + - grpcroutes + - referencegrants verbs: - get - list @@ -269,7 +366,29 @@ rules: - apiGroups: - velero.io resources: - - "*" + - backups + - restores + - schedules + - backupstoragelocations + - volumesnapshotlocations + - podvolumebackups + - podvolumerestores + - downloadrequests + - deletebackuprequests + - serverstatusrequests + verbs: + - get + - list + - watch +{{- end }} +{{- if .Values.crdPermissions.externalSecrets }} + - apiGroups: + - external-secrets.io + resources: + - externalsecrets + - secretstores + - clustersecretstores + - clusterexternalsecrets verbs: - get - list diff --git a/helm/holmes/values.yaml b/helm/holmes/values.yaml index 4edc4a905b..1c5ea5d294 100644 --- a/helm/holmes/values.yaml +++ b/helm/holmes/values.yaml @@ -39,7 +39,8 @@ crdPermissions: crossplane: true istio: true gatewayApi: true - velero: true + velero: true + externalSecrets: true enablePostProcessing: false postProcessingPrompt: "builtin://generic_post_processing.jinja2" From 70d1fefce65f2ec1aef76b86eea2a7960f824a17 Mon Sep 17 00:00:00 2001 From: moshemorad Date: Fri, 26 Dec 2025 00:52:31 +0200 Subject: [PATCH 23/27] Add health check to tcp connections (#1234) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ## Summary by CodeRabbit * **New Features** * Configurable LLM request timeout (env var; default 600s). * Optional TCP keepalive support with configurable idle, interval, and probe count. * Keepalive can be enabled/disabled at runtime via an environment flag. * **Chores** * Runtime now applies keepalive patching when enabled. ✏️ Tip: You can customize this high-level summary in your review settings. --------- Signed-off-by: Mohse Morad Signed-off-by: Tomer Keshet Signed-off-by: Codex Signed-off-by: Roi Glinik Signed-off-by: drfaust92 Co-authored-by: Tomer Co-authored-by: Natan Yellin Co-authored-by: Roi Glinik Co-authored-by: Ilia Lazebnik Co-authored-by: arik Signed-off-by: Filip Grebowski --- holmes/common/env_vars.py | 7 +++++++ holmes/core/llm.py | 2 ++ holmes/utils/connection_utils.py | 31 +++++++++++++++++++++++++++++++ server.py | 6 +++++- 4 files changed, 45 insertions(+), 1 deletion(-) create mode 100644 holmes/utils/connection_utils.py diff --git a/holmes/common/env_vars.py b/holmes/common/env_vars.py index 861b5685d2..e2629bf0da 100644 --- a/holmes/common/env_vars.py +++ b/holmes/common/env_vars.py @@ -113,3 +113,10 @@ def load_bool(env_var, default: Optional[bool]) -> Optional[bool]: ) SSE_READ_TIMEOUT = float(os.environ.get("SSE_READ_TIMEOUT", "120")) + +LLM_REQUEST_TIMEOUT = float(os.environ.get("LLM_REQUEST_TIMEOUT", "600")) + +ENABLE_CONNECTION_KEEPALIVE = load_bool("ENABLE_CONNECTION_KEEPALIVE", False) +KEEPALIVE_IDLE = int(os.environ.get("KEEPALIVE_IDLE", 2)) +KEEPALIVE_INTVL = int(os.environ.get("KEEPALIVE_INTVL", 2)) +KEEPALIVE_CNT = int(os.environ.get("KEEPALIVE_CNT", 5)) diff --git a/holmes/core/llm.py b/holmes/core/llm.py index 88d842dede..ad89c69e8d 100644 --- a/holmes/core/llm.py +++ b/holmes/core/llm.py @@ -21,6 +21,7 @@ from holmes.common.env_vars import ( FALLBACK_CONTEXT_WINDOW_SIZE, + LLM_REQUEST_TIMEOUT, LOAD_ALL_ROBUSTA_MODELS, REASONING_EFFORT, ROBUSTA_AI, @@ -421,6 +422,7 @@ def completion( drop_params=drop_params, allowed_openai_params=allowed_openai_params, stream=stream, + timeout=LLM_REQUEST_TIMEOUT, **tools_args, **self.args, ) diff --git a/holmes/utils/connection_utils.py b/holmes/utils/connection_utils.py new file mode 100644 index 0000000000..d479aa9365 --- /dev/null +++ b/holmes/utils/connection_utils.py @@ -0,0 +1,31 @@ +import socket +import logging + +from holmes.common.env_vars import KEEPALIVE_IDLE, KEEPALIVE_INTVL, KEEPALIVE_CNT + + +def patch_socket_create_connection( + idle: int = KEEPALIVE_IDLE, + intvl: int = KEEPALIVE_INTVL, + cnt: int = KEEPALIVE_CNT, +) -> None: + orig = socket.create_connection + + def new_create_connection(address, timeout=None, source_address=None, **kwargs): + logging.debug( + f"Creating patched connection to {address} with timeout {timeout} and source address {source_address}" + ) + s = orig(address, timeout=timeout, source_address=source_address, **kwargs) + s.setsockopt(socket.SOL_SOCKET, socket.SO_KEEPALIVE, 1) + + # Linux-only tuning (these attrs won't exist on macOS/Windows) + if hasattr(socket, "TCP_KEEPIDLE"): + s.setsockopt(socket.IPPROTO_TCP, socket.TCP_KEEPIDLE, int(idle)) + if hasattr(socket, "TCP_KEEPINTVL"): + s.setsockopt(socket.IPPROTO_TCP, socket.TCP_KEEPINTVL, int(intvl)) + if hasattr(socket, "TCP_KEEPCNT"): + s.setsockopt(socket.IPPROTO_TCP, socket.TCP_KEEPCNT, int(cnt)) + return s + + logging.info("Patching socket.create_connection to force keepalive") + socket.create_connection = new_create_connection diff --git a/server.py b/server.py index 7e8bbe2cbc..cd0c7654db 100644 --- a/server.py +++ b/server.py @@ -18,6 +18,7 @@ from holmes import get_version, is_official_release from holmes.core import investigation +from holmes.utils.connection_utils import patch_socket_create_connection from holmes.utils.holmes_status import update_holmes_status_in_db import logging import uvicorn @@ -29,6 +30,7 @@ from fastapi.responses import StreamingResponse from holmes.utils.stream import stream_investigate_formatter, stream_chat_formatter from holmes.common.env_vars import ( + ENABLE_CONNECTION_KEEPALIVE, HOLMES_HOST, HOLMES_PORT, HOLMES_POST_PROCESSING_PROMPT, @@ -86,6 +88,9 @@ def init_logging(): init_logging() + +if ENABLE_CONNECTION_KEEPALIVE: + patch_socket_create_connection() config = Config.load_from_env() dal = config.dal @@ -134,7 +139,6 @@ def sync_before_server_start(): app = FastAPI() - if LOG_PERFORMANCE: @app.middleware("http") From f1e121eb79ce6df20ad65d87b52dfa03864de5ab Mon Sep 17 00:00:00 2001 From: arik Date: Fri, 26 Dec 2025 07:03:22 +0200 Subject: [PATCH 24/27] Jq memory (#1232) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ## Summary by CodeRabbit * **New Features** * Added kubernetes_tabular_query for extracting and optionally filtering fields from Kubernetes resources with built-in tabular summarization. * **Enhancements** * kubernetes_jq_query now processes large responses via paginated, chunked batches with per-batch progress and final summaries. * kubernetes_count upgraded to chunked/batch processing with progress reporting and clearer aggregated results. * Improved summarization prompts and input thresholds for more reliable summaries and error/continuation handling. ✏️ Tip: You can customize this high-level summary in your review settings. --------- Signed-off-by: Arik Alon Signed-off-by: Filip Grebowski --- holmes/plugins/toolsets/kubernetes.yaml | 238 +++++++++++++++++++++--- 1 file changed, 208 insertions(+), 30 deletions(-) diff --git a/holmes/plugins/toolsets/kubernetes.yaml b/holmes/plugins/toolsets/kubernetes.yaml index 950825889b..038d446fd5 100644 --- a/holmes/plugins/toolsets/kubernetes.yaml +++ b/holmes/plugins/toolsets/kubernetes.yaml @@ -89,12 +89,97 @@ toolsets: - name: "kubernetes_jq_query" user_description: "Query Kubernetes Resources: kubectl get {{kind}} --all-namespaces -o json | jq -r {{jq_expr}}" description: > - Use kubectl to get json for all resources of a specific kind pipe the results to jq to filter them. Do not worry about escaping the jq_expr it will be done by the system on an unescaped expression that you give. e.g. give an expression like .items[] | .spec.containers[].image | select(test("^gcr.io/") | not) - command: kubectl get {{ kind }} --all-namespaces -o json | jq -r {{ jq_expr }} + Use kubectl to get json for all resources of a specific kind and filter with jq. + Do not worry about escaping the jq_expr - it will be done by the system. + Example: .items[] | .spec.containers[].image | select(test("^gcr.io/") | not) + script: | + #!/bin/bash + + echo "Executing paginated query for {{ kind }} resources..." + echo "Expression: {{ jq_expr }}" + echo "---" + + # Get the API path for the resource kind using kubectl and jq + API_INFO=$(kubectl api-resources --no-headers -o wide | grep "^{{ kind }}" | head -1) + + if [ -z "$API_INFO" ]; then + echo "Error: Unable to find resource kind '{{ kind }}'" >&2 + exit 1 + fi + + # Parse API info using read and bash string manipulation + IFS=' ' read -r RESOURCE_NAME SHORTNAME API_VERSION NAMESPACED REST <<< "$API_INFO" + + if [ -z "$API_VERSION" ] || [ -z "$RESOURCE_NAME" ]; then + echo "Error: Unable to parse API info for resource kind '{{ kind }}'" >&2 + exit 1 + fi + + # Build API path + if [[ "$API_VERSION" == "v1" ]]; then + API_PATH="/api/v1/${RESOURCE_NAME}" + else + API_PATH="/apis/${API_VERSION}/${RESOURCE_NAME}" + fi + + # Process resources in chunks using API pagination + LIMIT=500 # Process 500 items at a time + CONTINUE="" + PROCESSED=0 + TOTAL_MATCHES=0 + + while true; do + # Build API query with limit and continue token + if [ -z "$CONTINUE" ]; then + # First request - get from all namespaces + QUERY="${API_PATH}?limit=${LIMIT}" + else + # Subsequent requests with continue token + QUERY="${API_PATH}?limit=${LIMIT}&continue=${CONTINUE}" + fi + + OUTPUT=$(kubectl get --raw "$QUERY" 2>&1) + exit_code=$? + + if [ $exit_code -ne 0 ]; then + echo "Error: $OUTPUT" >&2 + exit $exit_code + fi + + ITEMS_COUNT=$(echo "$OUTPUT" | jq '.items | length') + + MATCHES=$(echo "$OUTPUT" | jq -r {{ jq_expr }} 2>&1) + jq_exit=$? + if [ $jq_exit -ne 0 ]; then + echo "Error: jq expression failed: $MATCHES" >&2 + exit $jq_exit + fi + + if [ "$ITEMS_COUNT" -gt 0 ]; then + if [ -n "$MATCHES" ]; then + echo "$MATCHES" + MATCH_COUNT=$(echo "$MATCHES" | grep -c . || true) + TOTAL_MATCHES=$((TOTAL_MATCHES + MATCH_COUNT)) + fi + + PROCESSED=$((PROCESSED + ITEMS_COUNT)) + + echo "Processed $PROCESSED items, found $TOTAL_MATCHES matches so far..." >&2 + fi + + CONTINUE=$(echo "$OUTPUT" | jq -r '.metadata.continue // empty') + + if [ -z "$CONTINUE" ]; then + break + fi + done + + echo "---" >&2 + echo "Total items processed: $PROCESSED, matches found: $TOTAL_MATCHES" >&2 transformers: - name: llm_summarize config: - input_threshold: 1000 + input_threshold: 10000 prompt: | Summarize this jq query output focusing on: - Key patterns and commonalities in the data @@ -106,6 +191,41 @@ toolsets: - Be concise: aim for ≤ 50% of the original text; prioritize aggregates and actionable outliers - Include grep-ready keys/values; avoid repeating entire objects or unchanged defaults + - name: "kubernetes_tabular_query" + user_description: "Tabular output of specific fields: kubectl get {{kind}} --all-namespaces -o custom-columns={{columns}}" + description: > + Extract specific fields from Kubernetes resources in tabular format with optional filtering. + Memory-efficient way to query large clusters - only requested fields are transmitted. + Column specification format: HEADER:FIELD_PATH,HEADER2:FIELD_PATH2,... + + Optional filtering parameter: + - filter_pattern: Pattern to match in any column (supports grep regex) + + Examples: + - Basic fields: NAME:.metadata.name,STATUS:.status.phase,NODE:.spec.nodeName + - Filter by status: filter_pattern="Running" + - Filter out lines with : filter_pattern="-v ''" + - Nested fields: CREATED:.metadata.creationTimestamp,IMAGE:.spec.containers[0].image + - Array fields: LABELS:.metadata.labels,PORTS:.spec.ports[*].port + + Note: Output is tabular text with column headers. Filtering works on the entire line. + Note: not allowed characters are: ' / ; and newline + command: kubectl get {{ kind }} --all-namespaces -o custom-columns='{{ columns }}'{% if filter_pattern %} | (head -n 1; tail -n +2 | grep {{ filter_pattern }}){% endif %} + transformers: + - name: llm_summarize + config: + input_threshold: 10000 + prompt: | + Summarize this tabular output focusing on: + - Key patterns and trends in the data + - Resources that need attention (errors, pending, failures) + - Group similar items into aggregate descriptions + - Highlight outliers or unusual values + - Mention specific resource names only for problematic items + - Provide counts and distributions where relevant + - Be concise: aim for ≤ 50% of the original size + - Keep output actionable and focused on anomalies + - name: "kubernetes_count" user_description: "Count Kubernetes Resources: kubectl get {{kind}} --all-namespaces -o json | jq -c -r {{ jq_expr }}" description: > @@ -115,43 +235,101 @@ toolsets: Do not worry about escaping the jq_expr it will be done by the system on an unescaped expression that you give. e.g. give an expression like .items[] | select(.spec.containers[].image | test("^gcr.io/") | not) | .metadata.name script: | + #!/bin/bash + echo "Command executed: kubectl get {{ kind }} --all-namespaces -o json | jq -c -r {{ jq_expr }}" echo "---" - # Execute the command and capture both stdout and stderr separately - temp_error=$(mktemp) - matches=$(kubectl get {{ kind }} --all-namespaces -o json 2>"$temp_error" | jq -c -r {{ jq_expr }} 2>>"$temp_error") - exit_code=$? - error_output=$(cat "$temp_error") - rm -f "$temp_error" - - if [ $exit_code -ne 0 ]; then - echo "Error executing command (exit code: $exit_code):" - echo "$error_output" - exit $exit_code + # Get the API path for the resource kind using kubectl + API_INFO=$(kubectl api-resources --no-headers -o wide | grep "^{{ kind }}" | head -1) + + if [ -z "$API_INFO" ]; then + echo "Error: Unable to find resource kind '{{ kind }}'" >&2 + exit 1 + fi + + # Parse API info using read and bash string manipulation + IFS=' ' read -r RESOURCE_NAME SHORTNAME API_VERSION NAMESPACED REST <<< "$API_INFO" + + if [ -z "$API_VERSION" ] || [ -z "$RESOURCE_NAME" ]; then + echo "Error: Unable to parse API info for resource kind '{{ kind }}'" >&2 + exit 1 + fi + + # Build API path + if [[ "$API_VERSION" == "v1" ]]; then + API_PATH="/api/v1/${RESOURCE_NAME}" else - # Show any stderr warnings even if command succeeded - if [ -n "$error_output" ]; then - echo "Warnings/stderr output:" - echo "$error_output" - echo "---" - fi + API_PATH="/apis/${API_VERSION}/${RESOURCE_NAME}" + fi + + # Process resources in chunks using API pagination + LIMIT=500 + CONTINUE="" + ALL_MATCHES="" + BATCH_NUM=0 + TOTAL_PROCESSED=0 + + while true; do + BATCH_NUM=$((BATCH_NUM + 1)) - # Filter out empty lines for accurate count - filtered_matches=$(echo "$matches" | grep -v '^$' | grep -v '^null$') - if [ -z "$filtered_matches" ]; then - count=0 + if [ -z "$CONTINUE" ]; then + QUERY="${API_PATH}?limit=${LIMIT}" else - count=$(echo "$filtered_matches" | wc -l) + QUERY="${API_PATH}?limit=${LIMIT}&continue=${CONTINUE}" + fi + + OUTPUT=$(kubectl get --raw "$QUERY" 2>&1) + exit_code=$? + + if [ $exit_code -ne 0 ]; then + echo "Error: $OUTPUT" >&2 + exit $exit_code + fi + + ITEMS_COUNT=$(echo "$OUTPUT" | jq '.items | length') + TOTAL_PROCESSED=$((TOTAL_PROCESSED + ITEMS_COUNT)) + + BATCH_MATCHES=$(echo "$OUTPUT" | jq -c -r {{ jq_expr }} 2>&1) + jq_exit=$? + if [ $jq_exit -ne 0 ]; then + echo "Error: jq expression failed: $BATCH_MATCHES" >&2 + exit $jq_exit + fi + + if [ -n "$BATCH_MATCHES" ]; then + if [ -z "$ALL_MATCHES" ]; then + ALL_MATCHES="$BATCH_MATCHES" + else + ALL_MATCHES="$ALL_MATCHES"$'\n'"$BATCH_MATCHES" + fi fi - preview=$(echo "$filtered_matches" | head -n 10 | cut -c 1-200 | nl) - echo "$count results" - echo "---" - echo "A *preview* of results is shown below (up to 10 results, up to 200 chars):" - echo "$preview" + CONTINUE=$(echo "$OUTPUT" | jq -r '.metadata.continue // empty') + if [ -z "$CONTINUE" ]; then + break + fi + + echo "Processed batch $BATCH_NUM ($TOTAL_PROCESSED items so far)..." >&2 + done + + # Now process the collected matches + filtered_matches=$(echo "$ALL_MATCHES" | grep -v '^$' | grep -v '^null$') + if [ -z "$filtered_matches" ]; then + count=0 + preview="" + else + count=$(echo "$filtered_matches" | wc -l) + preview=$(echo "$filtered_matches" | head -n 10 | cut -c 1-200 | nl) fi + echo "$count results" + echo "---" + echo "A *preview* of results is shown below (up to 10 results, up to 200 chars):" + echo "$preview" + echo "---" + echo "Total items processed: $TOTAL_PROCESSED" >&2 + # NOTE: this is only possible for probes with a healthz endpoint - we do this to avoid giving the LLM generic # http GET capabilities which are more powerful than we want to expose #- name: "check_liveness_probe" From beaf66c1421cd39b23704926dfc696175b1cac02 Mon Sep 17 00:00:00 2001 From: Natan Yellin Date: Fri, 26 Dec 2025 08:28:45 +0200 Subject: [PATCH 25/27] fix docker image notifications (#1245) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit They're currently being attached to files and not to the PR itself ## Summary by CodeRabbit * **Chores** * Workflow now triggers on pull requests (opened/synchronize/reopened) and cancels in-progress runs for the same head branch. * Reduces repo contents permission from write to read while retaining identity and PR permissions. * Captures build start time and short commit SHA (duplicate retrieval removed). * Posts/updates a "build started" comment on the PR and consolidates all comment operations to PR-based endpoints. * Always posts a final PR comment with computed build duration and detailed success/failure messaging. ✏️ Tip: You can customize this high-level summary in your review settings. --------- Signed-off-by: Robusta Runner Signed-off-by: Filip Grebowski --- .github/workflows/docker-dev-images.yaml | 276 +++++++++++++++-------- 1 file changed, 184 insertions(+), 92 deletions(-) diff --git a/.github/workflows/docker-dev-images.yaml b/.github/workflows/docker-dev-images.yaml index 2e1e85286f..9791abdf36 100644 --- a/.github/workflows/docker-dev-images.yaml +++ b/.github/workflows/docker-dev-images.yaml @@ -1,22 +1,125 @@ -name: Docker Build on Commit +name: Docker Build on PR on: - push: - branches: - - '**' # Trigger on push to any branch + pull_request: + types: [opened, synchronize, reopened] + +# Cancel in-progress runs for the same PR +concurrency: + group: docker-build-${{ github.head_ref }} + cancel-in-progress: true jobs: build: runs-on: ubuntu-latest permissions: - contents: 'write' + contents: 'read' id-token: 'write' pull-requests: 'write' steps: - uses: actions/checkout@v4 + - name: Get short SHA + id: short_sha + run: echo "sha_short=$(git rev-parse --short HEAD)" >> $GITHUB_OUTPUT + + - name: Record start time + id: start_time + run: echo "start=$(date +%s)" >> $GITHUB_OUTPUT + + - name: Comment build started + uses: actions/github-script@v7 + with: + script: | + const owner = context.repo.owner; + const repo = context.repo.repo; + const shortSha = `${{ steps.short_sha.outputs.sha_short }}`; + const runUrl = `https://github.com/${owner}/${repo}/actions/runs/${{ github.run_id }}`; + const marker = ''; + const registryLink = 'https://console.cloud.google.com/artifacts/docker/robusta-development/us-central1/temporary-builds/holmes?project=robusta-development'; + + const issue_number = context.payload.pull_request.number; + + const existingComments = await github.paginate(github.rest.issues.listComments, { + owner, + repo, + issue_number, + }); + const existing = existingComments.find(c => c.body?.includes(marker)); + + // Extract previous image tag from existing comment if present + let previousImageSection = ''; + if (existing) { + const tagMatch = existing.body.match(/holmes:([a-f0-9]{7})/); + if (tagMatch && tagMatch[1] !== shortSha) { + const prevSha = tagMatch[1]; + const prevTag = `us-central1-docker.pkg.dev/robusta-development/temporary-builds/holmes:${prevSha}`; + previousImageSection = [ + '', + '---', + `📦 **Previous image (\`${prevSha}\`):**`, + `- [${prevTag}](${registryLink})`, + '', + '
', + '📋 Copy commands', + '', + '⚠️ Temporary images are deleted after 30 days. Copy to a permanent registry before using them:', + '```bash', + 'gcloud auth configure-docker us-central1-docker.pkg.dev', + `docker pull ${prevTag}`, + `docker tag ${prevTag} me-west1-docker.pkg.dev/robusta-development/development/holmes-dev:${prevSha}`, + `docker push me-west1-docker.pkg.dev/robusta-development/development/holmes-dev:${prevSha}`, + '```', + '', + 'Patch Helm values in one line (choose the chart you use):', + '', + '**HolmesGPT chart:**', + '```bash', + 'helm upgrade --install holmesgpt ./helm/holmes \\\\', + ' --set registry=me-west1-docker.pkg.dev/robusta-development/development \\\\', + ` --set image=holmes-dev:${prevSha}`, + '```', + '', + '**Robusta wrapper chart:**', + '```bash', + 'helm upgrade --install robusta robusta/robusta \\\\', + ' --reuse-values \\\\', + ' --set holmes.registry=me-west1-docker.pkg.dev/robusta-development/development \\\\', + ` --set holmes.image=holmes-dev:${prevSha}`, + '```', + '
', + ].join('\n'); + } + } + + const message = [ + marker, + `🔨 **Building Docker image for \`${shortSha}\`...**`, + '', + `[View build logs](${runUrl})`, + previousImageSection, + ].join('\n'); + + if (existing) { + await github.rest.issues.updateComment({ + owner, + repo, + comment_id: existing.id, + body: message, + }); + core.info(`Updated existing PR comment #${existing.id}`); + } else { + await github.rest.issues.createComment({ + owner, + repo, + issue_number, + body: message, + }); + core.info(`Commented on PR #${issue_number}`); + } + - uses: google-github-actions/auth@v2 with: project_id: 'robusta-development' @@ -33,10 +136,6 @@ jobs: - name: Set up Docker Buildx uses: docker/setup-buildx-action@v3 - - name: Get short SHA - id: short_sha - run: echo "sha_short=$(git rev-parse --short HEAD)" >> $GITHUB_OUTPUT - - name: Build and push Docker image uses: docker/build-push-action@v6 with: @@ -56,106 +155,99 @@ jobs: echo " us-central1-docker.pkg.dev/robusta-development/temporary-builds/holmes:${{ steps.short_sha.outputs.sha_short }}" - name: Comment with image details + if: always() uses: actions/github-script@v7 with: script: | const owner = context.repo.owner; const repo = context.repo.repo; - const branch = context.ref.replace('refs/heads/', ''); - const sha = context.sha; const shortSha = `${{ steps.short_sha.outputs.sha_short }}`; + const jobStatus = `${{ job.status }}`; + const runUrl = `https://github.com/${owner}/${repo}/actions/runs/${{ github.run_id }}`; + const marker = ''; + + // Calculate build duration + const startTime = parseInt(`${{ steps.start_time.outputs.start }}`); + const endTime = Math.floor(Date.now() / 1000); + const durationSecs = endTime - startTime; + const minutes = Math.floor(durationSecs / 60); + const seconds = durationSecs % 60; + const duration = minutes > 0 ? `${minutes}m ${seconds}s` : `${seconds}s`; + + let message; + if (jobStatus === 'success') { + const shortTag = `us-central1-docker.pkg.dev/robusta-development/temporary-builds/holmes:${shortSha}`; + const registryLink = 'https://console.cloud.google.com/artifacts/docker/robusta-development/us-central1/temporary-builds/holmes?project=robusta-development'; + + message = [ + marker, + `✅ **Docker image ready for \`${shortSha}\`** (built in ${duration})`, + '', + `- [${shortTag}](${registryLink})`, + '', + 'Use this tag to pull the image for testing.', + '', + '
', + '📋 Copy commands', + '', + '⚠️ Temporary images are deleted after 30 days. Copy to a permanent registry before using them:', + '```bash', + 'gcloud auth configure-docker us-central1-docker.pkg.dev', + `docker pull ${shortTag}`, + `docker tag ${shortTag} me-west1-docker.pkg.dev/robusta-development/development/holmes-dev:${shortSha}`, + `docker push me-west1-docker.pkg.dev/robusta-development/development/holmes-dev:${shortSha}`, + '```', + '', + 'Patch Helm values in one line (choose the chart you use):', + '', + '**HolmesGPT chart:**', + '```bash', + 'helm upgrade --install holmesgpt ./helm/holmes \\', + ' --set registry=me-west1-docker.pkg.dev/robusta-development/development \\', + ` --set image=holmes-dev:${shortSha}`, + '```', + '', + '**Robusta wrapper chart:**', + '```bash', + 'helm upgrade --install robusta robusta/robusta \\', + ' --reuse-values \\', + ' --set holmes.registry=me-west1-docker.pkg.dev/robusta-development/development \\', + ` --set holmes.image=holmes-dev:${shortSha}`, + '```', + '
', + ].join('\n'); + } else { + message = [ + marker, + `❌ **Docker build failed for \`${shortSha}\`** (after ${duration})`, + '', + `[View build logs](${runUrl})`, + ].join('\n'); + } - const shortTag = `us-central1-docker.pkg.dev/robusta-development/temporary-builds/holmes:${shortSha}`; - const registryLink = 'https://console.cloud.google.com/artifacts/docker/robusta-development/us-central1/temporary-builds/holmes?project=robusta-development'; - const marker = 'Dev Docker images are ready for this commit:'; - - const message = [ - marker, - '', - `- [${shortTag}](${registryLink})`, - '', - 'Use this tag to pull the image for testing.', - '', - '⚠️ Temporary images are deleted after 30 days. Copy to a permanent registry before using them:', - '```bash', - 'gcloud auth configure-docker us-central1-docker.pkg.dev', - `docker pull ${shortTag}`, - `docker tag ${shortTag} me-west1-docker.pkg.dev/robusta-development/development/holmes-dev:${shortSha}`, - `docker push me-west1-docker.pkg.dev/robusta-development/development/holmes-dev:${shortSha}`, - '```', - '', - 'Patch Helm values in one line (choose the chart you use):', - '- HolmesGPT chart:', - '```bash', - 'helm upgrade --install holmesgpt ./helm/holmes \\', - ' --set registry=me-west1-docker.pkg.dev/robusta-development/development \\', - ` --set image=holmes-dev:${shortSha}`, - '```', - '- Robusta wrapper chart:', - '```bash', - 'helm upgrade --install robusta robusta/robusta \\', - ' --reuse-values \\', - ' --set holmes.registry=me-west1-docker.pkg.dev/robusta-development/development \\', - ` --set holmes.image=holmes-dev:${shortSha}`, - '```', - ].join('\n'); + const issue_number = context.payload.pull_request.number; - const prs = await github.paginate(github.rest.pulls.list, { + const existingComments = await github.paginate(github.rest.issues.listComments, { owner, repo, - state: 'open', - head: `${owner}:${branch}`, + issue_number, }); + const existing = existingComments.find(c => c.body?.includes(marker)); - if (prs.length > 0) { - const issue_number = prs[0].number; - const existingComments = await github.paginate(github.rest.issues.listComments, { + if (existing) { + await github.rest.issues.updateComment({ owner, repo, - issue_number, + comment_id: existing.id, + body: message, }); - const existing = existingComments.find(c => c.body?.includes(marker)); - - if (existing) { - await github.rest.issues.updateComment({ - owner, - repo, - comment_id: existing.id, - body: message, - }); - core.info(`Updated existing PR comment #${existing.id}`); - } else { - await github.rest.issues.createComment({ - owner, - repo, - issue_number, - body: message, - }); - core.info(`Commented on PR #${issue_number}`); - } + core.info(`Updated existing PR comment #${existing.id}`); } else { - const commitComments = await github.paginate(github.rest.repos.listCommentsForCommit, { + await github.rest.issues.createComment({ owner, repo, - commit_sha: sha, + issue_number, + body: message, }); - const existing = commitComments.find(c => c.body?.includes(marker)); - - if (existing) { - await github.rest.repos.updateCommitComment({ - owner, - repo, - comment_id: existing.id, - body: message, - }); - core.info(`Updated existing commit comment #${existing.id}`); - } else { - await github.rest.repos.createCommitComment({ - owner, - repo, - commit_sha: sha, - body: message, - }); - core.info('Commented on commit'); - } + core.info(`Commented on PR #${issue_number}`); } From cf5d99d0757f67ddf6c2500b861719365cc45a6a Mon Sep 17 00:00:00 2001 From: moshemorad Date: Fri, 26 Dec 2025 09:03:12 +0200 Subject: [PATCH 26/27] Fix oci helm push action (#1246) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ## Summary by CodeRabbit * **Chores** * Updated CI/CD pipeline configuration for Docker image and Helm chart deployment processes. **Note:** This release contains internal infrastructure updates with no user-facing changes. ✏️ Tip: You can customize this high-level summary in your review settings. Signed-off-by: Mohse Morad Signed-off-by: Filip Grebowski --- .github/workflows/build-docker-images.yaml | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/.github/workflows/build-docker-images.yaml b/.github/workflows/build-docker-images.yaml index 02ca00cb48..34f91adeef 100644 --- a/.github/workflows/build-docker-images.yaml +++ b/.github/workflows/build-docker-images.yaml @@ -88,5 +88,4 @@ jobs: - name: Push Helm chart to OCI registry run: | helm package helm/holmes - helm push holmes-${{github.ref_name}}.tgz oci://ghcr.io/${{ github.repository_owner }}/charts - + helm push holmes-${{github.ref_name}}.tgz oci://ghcr.io/holmesgpt/charts From 5b4742cf1e4de129bbfb4eb82d75fd7c20a00a2f Mon Sep 17 00:00:00 2001 From: Filip Grebowski Date: Sat, 27 Dec 2025 13:23:49 +0000 Subject: [PATCH 27/27] Corrected file names and titles Signed-off-by: Filip Grebowski --- docs/installation/.nav.yml | 2 +- ...stallation.md => backstage-integration.md} | 4 +--- docs/installation/k9s-installation.md | 19 +++++++------------ docs/installation/slack-installation.md | 14 +++++--------- docs/installation/ui-installation.md | 16 ++++++---------- mkdocs.yml | 6 +++--- 6 files changed, 23 insertions(+), 38 deletions(-) rename docs/installation/{backstage-installation.md => backstage-integration.md} (97%) diff --git a/docs/installation/.nav.yml b/docs/installation/.nav.yml index 2a7334e488..28995125bf 100644 --- a/docs/installation/.nav.yml +++ b/docs/installation/.nav.yml @@ -2,7 +2,7 @@ nav: - Install CLI: cli-installation.md - Install UI (3rd party): ui-installation.md - Install Slack Bot (3rd party): slack-installation.md - - Install Backstage (3rd party): backstage-installation.md + - Integration with Backstage (3rd party): backstage-integration.md - Install K9s: k9s-installation.md - Install Helm Chart: kubernetes-installation.md - Install Python SDK: python-installation.md diff --git a/docs/installation/backstage-installation.md b/docs/installation/backstage-integration.md similarity index 97% rename from docs/installation/backstage-installation.md rename to docs/installation/backstage-integration.md index 4a0064f40f..83bafaaf0e 100644 --- a/docs/installation/backstage-installation.md +++ b/docs/installation/backstage-integration.md @@ -1,4 +1,4 @@ -# Install Backstage (third party) +# Integration with Backstage (3rd party) There’s now a third-party integration that brings **HolmesGPT** by **Robusta.dev** directly into Backstage. The integration is currently available in **closed beta**. @@ -27,5 +27,3 @@ Below is a short walkthrough of the HolmesGPT–Backstage integration. It demons If you’re already running Backstage and want to bring AI-driven diagnostics into your developer portal, you can request access to the closed beta by contacting the Robusta.dev team via [email](mailto:arik@robusta.dev) or [slack](https://robustacommunity.slack.com). You can also use [this form](https://robusta-dev.typeform.com/to/xhyQYgLx) to register your interest. - - diff --git a/docs/installation/k9s-installation.md b/docs/installation/k9s-installation.md index f6f50be66a..25d5971da4 100644 --- a/docs/installation/k9s-installation.md +++ b/docs/installation/k9s-installation.md @@ -1,15 +1,13 @@ -# Install K9s +# Install K9s Plugin -## K9s Plugin - -Integrate HolmesGPT into your [K9s](https://github.com/derailed/k9s){:target="_blank"} Kubernetes terminal for instant analysis. +Integrate HolmesGPT into your [K9s](https://github.com/derailed/k9s){:target="\_blank"} Kubernetes terminal for instant analysis. ![K9s Demo](../assets/K9sDemo.gif) ### Prerequisites -- **K9s must be installed** - See the [K9s installation guide](https://github.com/derailed/k9s#installation){:target="_blank"} -- **HolmesGPT CLI and API key** - Follow the [CLI Installation Guide](cli-installation.md) to install Holmes and configure your AI provider +- **K9s must be installed** - See the [K9s installation guide](https://github.com/derailed/k9s#installation){:target="\_blank"} +- **HolmesGPT CLI and API key** - Follow the [CLI Installation Guide](cli-installation.md) to install Holmes and configure your AI provider ### Plugin Options @@ -63,7 +61,6 @@ Integrate HolmesGPT into your [K9s](https://github.com/derailed/k9s){:target="_b ??? note "Advanced Plugin (Shift + Q) - Interactive plugin with custom questions" - Add to your K9s plugins configuration file: - **Linux**: `~/.config/k9s/plugins.yaml` or `~/.k9s/plugins.yaml` @@ -131,8 +128,6 @@ Integrate HolmesGPT into your [K9s](https://github.com/derailed/k9s){:target="_b ## Need Help? -- **[Join our Slack](https://cloud-native.slack.com/archives/C0A1SPQM5PZ){:target="_blank"}** - Get help from the community -- **[Request features on GitHub](https://github.com/HolmesGPT/holmesgpt/issues){:target="_blank"}** - Suggest improvements or report bugs -- **[Troubleshooting guide](../reference/troubleshooting.md)** - Common issues and solutions - - +- **[Join our Slack](https://cloud-native.slack.com/archives/C0A1SPQM5PZ){:target="\_blank"}** - Get help from the community +- **[Request features on GitHub](https://github.com/HolmesGPT/holmesgpt/issues){:target="\_blank"}** - Suggest improvements or report bugs +- **[Troubleshooting guide](../reference/troubleshooting.md)** - Common issues and solutions diff --git a/docs/installation/slack-installation.md b/docs/installation/slack-installation.md index c3e326101f..749ccb0e99 100644 --- a/docs/installation/slack-installation.md +++ b/docs/installation/slack-installation.md @@ -1,6 +1,4 @@ -# Install Slack Bot (third party) - -## Slack Bot (Robusta) +# Install Robusta Slack Bot (3rd party) First install Robusta SaaS, then tag HolmesGPT in any Slack message for instant analysis. @@ -8,12 +6,10 @@ First install Robusta SaaS, then tag HolmesGPT in any Slack message for instant ### Setup Slack Bot -[![Watch Slack Bot Demo](https://cdn.loom.com/sessions/thumbnails/7a60a42e854e45368e9b7f9d3c36ae5f-65bd123629db6922-full-play.gif)](https://www.loom.com/share/7a60a42e854e45368e9b7f9d3c36ae5f?sid=bfed9efb-b607-416c-b481-c2a63d314a4b){:target="_blank"} +[![Watch Slack Bot Demo](https://cdn.loom.com/sessions/thumbnails/7a60a42e854e45368e9b7f9d3c36ae5f-65bd123629db6922-full-play.gif)](https://www.loom.com/share/7a60a42e854e45368e9b7f9d3c36ae5f?sid=bfed9efb-b607-416c-b481-c2a63d314a4b){:target="\_blank"} ## Need Help? -- **[Join our Slack](https://cloud-native.slack.com/archives/C0A1SPQM5PZ){:target="_blank"}** - Get help from the community -- **[Request features on GitHub](https://github.com/HolmesGPT/holmesgpt/issues){:target="_blank"}** - Suggest improvements or report bugs -- **[Troubleshooting guide](../reference/troubleshooting.md)** - Common issues and solutions - - +- **[Join our Slack](https://cloud-native.slack.com/archives/C0A1SPQM5PZ){:target="\_blank"}** - Get help from the community +- **[Request features on GitHub](https://github.com/HolmesGPT/holmesgpt/issues){:target="\_blank"}** - Suggest improvements or report bugs +- **[Troubleshooting guide](../reference/troubleshooting.md)** - Common issues and solutions diff --git a/docs/installation/ui-installation.md b/docs/installation/ui-installation.md index 97b68bfe4e..5d2fe7959f 100644 --- a/docs/installation/ui-installation.md +++ b/docs/installation/ui-installation.md @@ -1,6 +1,4 @@ -# Install UI (third party) - -## Web UI (Robusta) +# Install Web UI (3rd party) The fastest way to use HolmesGPT is via the managed Robusta SaaS platform. @@ -34,17 +32,15 @@ The fastest way to use HolmesGPT is via the managed Robusta SaaS platform. ### Get Started -1. **Sign up:** [platform.robusta.dev](https://platform.robusta.dev/signup/?utm_source=docs&utm_medium=holmesgpt-docs&utm_content=ui_installation_section){:target="_blank"} +1. **Sign up:** [platform.robusta.dev](https://platform.robusta.dev/signup/?utm_source=docs&utm_medium=holmesgpt-docs&utm_content=ui_installation_section){:target="\_blank"} 2. **Connect your cluster:** Follow the in-app wizard 3. **Ask Holmes:** Analyze alerts or troubleshoot issues !!! tip "Multiple AI Providers" - You can configure multiple AI models for users to choose from in the UI. See [Using Multiple Providers](../ai-providers/using-multiple-providers.md) for configuration details. +You can configure multiple AI models for users to choose from in the UI. See [Using Multiple Providers](../ai-providers/using-multiple-providers.md) for configuration details. ## Need Help? -- **[Join our Slack](https://cloud-native.slack.com/archives/C0A1SPQM5PZ){:target="_blank"}** - Get help from the community -- **[Request features on GitHub](https://github.com/HolmesGPT/holmesgpt/issues){:target="_blank"}** - Suggest improvements or report bugs -- **[Troubleshooting guide](../reference/troubleshooting.md)** - Common issues and solutions - - +- **[Join our Slack](https://cloud-native.slack.com/archives/C0A1SPQM5PZ){:target="\_blank"}** - Get help from the community +- **[Request features on GitHub](https://github.com/HolmesGPT/holmesgpt/issues){:target="\_blank"}** - Suggest improvements or report bugs +- **[Troubleshooting guide](../reference/troubleshooting.md)** - Common issues and solutions diff --git a/mkdocs.yml b/mkdocs.yml index 26e3a0bc24..9ffe88caca 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -14,9 +14,9 @@ nav: - Home: index.md - Installation: - "Install CLI": installation/cli-installation.md - - "Install UI (third party)": installation/ui-installation.md - - "Install Slack Bot (third party)": installation/slack-installation.md - - "Install Backstage (third party)": installation/backstage-installation.md + - "Install UI (3rd party)": installation/ui-installation.md + - "Install Slack Bot (3rd party)": installation/slack-installation.md + - "Integration with Backstage (3rd party)": installation/backstage-integration.md - "Install K9s": installation/k9s-installation.md - "Install Helm Chart": installation/kubernetes-installation.md - "Install Python SDK": installation/python-installation.md