diff --git a/holmes/config.py b/holmes/config.py index ee8c362b4d..fa3421e8ff 100644 --- a/holmes/config.py +++ b/holmes/config.py @@ -18,7 +18,11 @@ from holmes.core.tool_calling_llm import IssueInvestigator, ToolCallingLLM, ToolExecutor from holmes.core.toolset_manager import ToolsetManager from holmes.plugins.destinations.slack import SlackDestination -from holmes.plugins.runbooks import load_builtin_runbooks, load_runbooks_from_file +from holmes.plugins.runbooks import ( + load_builtin_runbooks, + load_runbook_catalog, + load_runbooks_from_file, +) from holmes.plugins.sources.github import GitHubSource from holmes.plugins.sources.jira import JiraServiceManagementSource, JiraSource from holmes.plugins.sources.opsgenie import OpsGenieSource @@ -230,6 +234,16 @@ def __get_cluster_name() -> Optional[str]: return None + @staticmethod + def get_runbook_catalog() -> str: + # TODO(mainred): besides the built-in runbooks, we need to allow the user to bring their own runbooks + runbook_catalog = load_runbook_catalog() + if runbook_catalog is not None: + return runbook_catalog.model_dump_json() + else: + logging.warning("Runbook catalog not found") + return json.dumps({"catalog": []}) + def create_console_tool_executor(self, dal: Optional[SupabaseDal]) -> ToolExecutor: """ Creates a ToolExecutor instance configured for CLI usage. This executor manages the available tools diff --git a/holmes/main.py b/holmes/main.py index cba0bd9fc9..19f6f079b1 100644 --- a/holmes/main.py +++ b/holmes/main.py @@ -26,7 +26,6 @@ from rich.markdown import Markdown from rich.rule import Rule - from holmes import get_version # type: ignore from holmes.config import ( DEFAULT_CONFIG_LOCATION, @@ -38,12 +37,12 @@ from holmes.core.resource_instruction import ResourceInstructionDocument from holmes.core.tool_calling_llm import LLMResult from holmes.core.tools import pretty_print_toolset_status +from holmes.interactive import run_interactive_loop from holmes.plugins.destinations import DestinationType from holmes.plugins.interfaces import Issue from holmes.plugins.prompts import load_and_render_prompt from holmes.plugins.sources.opsgenie import OPSGENIE_TEAM_INTEGRATION_KEY_HELP from holmes.utils.file_utils import write_json_file -from holmes.interactive import run_interactive_loop app = typer.Typer(add_completion=False, pretty_exceptions_show_locals=False) investigate_app = typer.Typer( @@ -320,6 +319,7 @@ def ask( ) template_context = { "toolsets": ai.tool_executor.toolsets, + "runbooks": config.get_runbook_catalog(), } system_prompt_rendered = load_and_render_prompt(system_prompt, template_context) # type: ignore diff --git a/holmes/plugins/prompts/_general_instructions.jinja2 b/holmes/plugins/prompts/_general_instructions.jinja2 index 997c6cdcf7..b13bf85005 100644 --- a/holmes/plugins/prompts/_general_instructions.jinja2 +++ b/holmes/plugins/prompts/_general_instructions.jinja2 @@ -75,4 +75,4 @@ Reminder: * That is different than - for example - fetching a pod's logs and seeing that the pod itself has permission errors. in that case, you explain say that permission errors are the cause of the problem and give details * Issues are a subset of findings. When asked about an issue or a finding and you have an id, use the tool `fetch_finding_by_id`. * For any question, try to make the answer specific to the user's cluster. -** For example, if asked to port forward, find out the app or pod port (kubectl decribe) and provide a port forward command specific to the user's question +** For example, if asked to port forward, find out the app or pod port (kubectl describe) and provide a port forward command specific to the user's question diff --git a/holmes/plugins/prompts/_runbook_instructions.jinja2 b/holmes/plugins/prompts/_runbook_instructions.jinja2 new file mode 100644 index 0000000000..2d4bba4b79 --- /dev/null +++ b/holmes/plugins/prompts/_runbook_instructions.jinja2 @@ -0,0 +1,17 @@ +{% if runbooks and runbooks.catalog|length > 0 %} +# Runbook Selection + +# Available Runbooks + +{%- for runbook in runbooks.catalog -%} + +description: {{ runbook.description }} +link: {{ runbook.link }} + +{%- endfor -%} + +ALWAYS try to find the runbooks that can provide troubleshooting instructions when the user describes an operational issue, debugging scenario, or asks for step‑by‑step troubleshooting. +To get the runbook details, use `fetch_runbook` tool by comparing the runbook description with the user prompt. +ALWAYS follow the steps described in the runbook. +If you decided not to follow one or more steps, ALWAYS explain why. +{%- endif -%} diff --git a/holmes/plugins/prompts/generic_ask.jinja2 b/holmes/plugins/prompts/generic_ask.jinja2 index 6630a2a39b..640cf9308b 100644 --- a/holmes/plugins/prompts/generic_ask.jinja2 +++ b/holmes/plugins/prompts/generic_ask.jinja2 @@ -11,6 +11,8 @@ Bias towards not asking the user for help if you can find the answer yourself. {% include '_general_instructions.jinja2' %} +{% include '_runbook_instructions.jinja2' %} + # Style guide * Reply with terse output. diff --git a/holmes/plugins/runbooks/README.md b/holmes/plugins/runbooks/README.md new file mode 100644 index 0000000000..217a6214b3 --- /dev/null +++ b/holmes/plugins/runbooks/README.md @@ -0,0 +1,22 @@ +# Runbooks + +Runbooks folder contains operational runbooks for the HolmesGPT project. Runbooks provide step-by-step instructions for common tasks, troubleshooting, and maintenance procedures related to the plugins in this directory. + +## Purpose + +- Standardize operational processes +- Enable quick onboarding for new team members +- Reduce downtime by providing clear troubleshooting steps + +## Structure + +### Structured Runbook + +Structured runbooks are designed for specific issues when conditions like issue name, id or source match, the corresponding instructions will be returned for investigation. +For example, the investigation in [kube-prometheus-stack.yaml](kube-prometheus-stack.yaml) will be returned when the issue to be investigated match either KubeSchedulerDown or KubeControllerManagerDown. +This runbook is mainly used for `holmes investigate` + +### Catalog + +Catalog specified in [catalog.json](catalog.json) contains a collection of runbooks written in markdown. +During runtime, LLM will compare the runbook description with the user question and return the most matched runbook for investigation. It's possible no runbook is returned for no match. diff --git a/holmes/plugins/runbooks/__init__.py b/holmes/plugins/runbooks/__init__.py index 9da78675f6..4b44e0db74 100644 --- a/holmes/plugins/runbooks/__init__.py +++ b/holmes/plugins/runbooks/__init__.py @@ -1,5 +1,8 @@ +import json +import logging import os import os.path +from datetime import date from pathlib import Path from typing import List, Optional, Pattern, Union @@ -9,6 +12,8 @@ THIS_DIR = os.path.abspath(os.path.dirname(__file__)) +CATALOG_FILE = "catalog.json" + class IssueMatcher(RobustaBaseConfig): issue_id: Optional[Pattern] = None # unique id @@ -48,3 +53,48 @@ def load_builtin_runbooks() -> List[Runbook]: path = os.path.join(THIS_DIR, filename) all_runbooks.extend(load_runbooks_from_file(path)) return all_runbooks + + +class RunbookCatalogEntry(BaseModel): + """ + RunbookCatalogEntry contains metadata about a runbook + Different from runbooks provided by Runbook class, this entry points to markdown file containing the runbook content. + """ + + update_date: date + description: str + link: str + + +class RunbookCatalog(BaseModel): + """ + RunbookCatalog is a collection of runbook entries, each entry contains metadata about the runbook. + The correct runbook can be selected from the list by comparing the description with the user question. + """ + + catalog: List[RunbookCatalogEntry] + + +def load_runbook_catalog() -> Optional[RunbookCatalog]: + dir_path = os.path.dirname(os.path.realpath(__file__)) + + catalogPath = os.path.join(dir_path, CATALOG_FILE) + if not os.path.isfile(catalogPath): + return None + try: + with open(catalogPath) as file: + catalog_dict = json.load(file) + return RunbookCatalog(**catalog_dict) + except json.JSONDecodeError as e: + logging.error(f"Error decoding JSON from {catalogPath}: {e}") + except Exception as e: + logging.error( + f"Unexpected error while loading runbook catalog from {catalogPath}: {e}" + ) + return None + + +def get_runbook_by_path(runbook_relative_path: str) -> str: + runbook_folder = os.path.dirname(os.path.realpath(__file__)) + runbook_path = os.path.join(runbook_folder, runbook_relative_path) + return runbook_path diff --git a/holmes/plugins/runbooks/catalog.json b/holmes/plugins/runbooks/catalog.json new file mode 100644 index 0000000000..7550db5592 --- /dev/null +++ b/holmes/plugins/runbooks/catalog.json @@ -0,0 +1,9 @@ +{ + "catalog": [ + { + "update_date": "2025-06-17", + "description": "Runbook to investigate DNS resolution issue on Kubernetes cluster", + "link": "networking/dns_troubleshooting_instructions.md" + } + ] +} diff --git a/holmes/plugins/runbooks/networking/dns_troubleshooting_instructions.md b/holmes/plugins/runbooks/networking/dns_troubleshooting_instructions.md new file mode 100644 index 0000000000..26e86da5fb --- /dev/null +++ b/holmes/plugins/runbooks/networking/dns_troubleshooting_instructions.md @@ -0,0 +1,66 @@ +# DNS Troubleshooting Guidelines (Kubernetes) + +## Goal +Your primary goal when using these tools is to diagnose DNS resolution issues within a Kubernetes cluster, focusing on identifying common problems like incorrect CoreDNS/kube-dns setup, network policies, or service discovery failures by strictly following the workflow for DNS diagnosis. + +* Use the tools to gather information about the DNS pods, services, and configuration. +* Clearly present the key findings from the tool outputs in your analysis. +* Instead of provide next steps to the user, you need to follow the troubleshoot guide to execute the steps. +* When getting pod logs, always try to get the log filter by log_filter toolset to filter out unnecessary logs by tool kubectl_logs_grep_no_match + +## Workflow for DNS Diagnosis + +1. **Check CoreDNS/kube-dns Pods:** + * Verify that the DNS pods (e.g., CoreDNS or kube-dns) are running in the `kube-system` namespace. + * Look for restarts or crashes in the DNS pods. + +2. **Examine DNS Service:** + * Ensure the DNS service is correctly defined: `kubectl get svc kube-dns -n kube-system` (or the equivalent for your DNS provider). + * Verify the ClusterIP of the DNS service and the ports (usually 53/UDP and 53/TCP). + +3. **Test DNS Resolution from a Pod:** + * Launch a debugging pod (e.g., using `busybox` or `nslookup` tools). + * **Inside the debug pod:** + * Check `/etc/resolv.conf`: + * The `nameserver` should point to the DNS service's ClusterIP. + * The `search` path should be appropriate for your namespaces (e.g., `your-namespace.svc.cluster.local svc.cluster.local cluster.local`). + * The `options` (like `ndots:5`) can affect resolution behavior. + * Attempt to resolve internal cluster names: + * A service in the same namespace (e.g., `myservice`). + * A service in a different namespace (e.g., `myservice.othernamespace`). + * A fully qualified domain name (FQDN) (e.g., `myservice.othernamespace.svc.cluster.local`). + * Attempt to resolve external names (e.g., `www.google.com`). + * Use tools like `nslookup ` or `dig ` for detailed query information. + +4. **Check NetworkPolicies:** + * If NetworkPolicies are in place, ensure they allow DNS traffic (to port 53 UDP/TCP) from your application pods to the DNS pods/service. + * List NetworkPolicies and Examine policies that might be affecting the source or destination pods. + +5. **Review CoreDNS Configuration (if applicable):** + * Inspect the CoreDNS ConfigMap: `kubectl get configmap coredns -n kube-system -o yaml`. + * Look for errors or misconfigurations in the Corefile (e.g., incorrect upstream resolvers, plugin issues). + * Inspect the customized CoreDNS ConfigMap: `kubectl get configmap coredns-custom -n kube-system -o yaml`. + * Look for errors or misconfigurations in the customizated CoreDNS config (e.g., incorrect upstream resolvers, plugin issues). + +6. **Check the DNS trace** + * Use findings from the DNS trace to pinpoint where DNS resolution is failing (e.g., query not reaching DNS server, invalid FQDN, or error response from DNS server). + * DNS Server should always respond to the requests from the client. Valid FQDN should return NOERROR, and invalid FQDN should return NXDOMAIN + +## Synthesize Findings +Based on the outputs from the above steps, describe the DNS issue clearly. For example: +* "DNS resolution for internal service 'myservice' is failing from pods in namespace 'app-ns'. The CoreDNS pods in `kube-system` are running but show 'connection refused' errors in their logs when trying to reach upstream resolvers." +* "Pods in namespace 'secure-ns' cannot resolve any hostnames. `/etc/resolv.conf` in these pods is missing the correct `nameserver` entry. This is likely due to a misconfiguration in the pod's `dnsPolicy` or the underlying node's DNS setup." +* "External DNS resolution is failing cluster-wide. The CoreDNS ConfigMap shows that the `forward` plugin is pointing to an incorrect upstream DNS server IP address." +* "DNS lookups for 'service-a.namespace-b' are timing out. A NetworkPolicy in 'namespace-b' is blocking egress traffic on port 53 to the kube-dns service." + +## Recommend Remediation Steps (Based on Docs) +* **CRITICAL:** ALWAYS refer to the official Kubernetes DNS debugging guide for detailed troubleshooting and solutions: + * Main guide: https://kubernetes.io/docs/tasks/administer-cluster/dns-debugging-resolution/ + * CoreDNS specific: https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/ (for CoreDNS customization which might be relevant) +* **DO NOT invent recovery procedures.** Your role is to diagnose and *point* to the correct documentation or standard procedures. +* Based on the findings, suggest which sections of the documentation are most relevant. + * If DNS pods are not running, guide towards checking pod deployment and node health. + * If `/etc/resolv.conf` is incorrect, point to sections on Pod `dnsPolicy` and `dnsConfig`. + * If NetworkPolicies are suspected, suggest reviewing policy definitions to allow DNS. + * If CoreDNS configuration seems problematic, refer to CoreDNS documentation and the Kubernetes guide on customizing it. + * If upstream DNS resolution is failing, suggest checking the upstream DNS servers and CoreDNS forward configuration. diff --git a/holmes/plugins/toolsets/__init__.py b/holmes/plugins/toolsets/__init__.py index c4b4d7e316..e47dc3816b 100644 --- a/holmes/plugins/toolsets/__init__.py +++ b/holmes/plugins/toolsets/__init__.py @@ -9,13 +9,13 @@ import holmes.utils.env as env_utils from holmes.core.supabase_dal import SupabaseDal from holmes.core.tools import Toolset, ToolsetType, ToolsetYamlFromConfig, YAMLToolset +from holmes.plugins.toolsets.bash.bash_toolset import BashExecutorToolset from holmes.plugins.toolsets.coralogix.toolset_coralogix_logs import ( CoralogixLogsToolset, ) from holmes.plugins.toolsets.datadog import DatadogToolset from holmes.plugins.toolsets.git import GitToolset from holmes.plugins.toolsets.grafana.toolset_grafana import GrafanaToolset -from holmes.plugins.toolsets.bash.bash_toolset import BashExecutorToolset from holmes.plugins.toolsets.grafana.toolset_grafana_loki import GrafanaLokiToolset from holmes.plugins.toolsets.grafana.toolset_grafana_tempo import GrafanaTempoToolset from holmes.plugins.toolsets.internet.internet import InternetToolset @@ -29,6 +29,7 @@ from holmes.plugins.toolsets.prometheus.prometheus import PrometheusToolset from holmes.plugins.toolsets.rabbitmq.toolset_rabbitmq import RabbitMQToolset from holmes.plugins.toolsets.robusta.robusta import RobustaToolset +from holmes.plugins.toolsets.runbook.runbook_fetcher import RunbookToolset THIS_DIR = os.path.abspath(os.path.dirname(__file__)) @@ -70,6 +71,7 @@ def load_python_toolsets(dal: Optional[SupabaseDal]) -> List[Toolset]: RabbitMQToolset(), GitToolset(), BashExecutorToolset(), + RunbookToolset(), ] return toolsets diff --git a/holmes/plugins/toolsets/runbook/__init__.py b/holmes/plugins/toolsets/runbook/__init__.py new file mode 100644 index 0000000000..e69de29bb2 diff --git a/holmes/plugins/toolsets/runbook/runbook_fetcher.py b/holmes/plugins/toolsets/runbook/runbook_fetcher.py new file mode 100644 index 0000000000..56b78b8be6 --- /dev/null +++ b/holmes/plugins/toolsets/runbook/runbook_fetcher.py @@ -0,0 +1,78 @@ +import logging +from typing import Any, Dict + +from holmes.core.tools import ( + StructuredToolResult, + Tool, + ToolParameter, + ToolResultStatus, + Toolset, + ToolsetTag, +) +from holmes.plugins.runbooks import get_runbook_by_path + + +# TODO(mainred): currently we support fetch runbooks hosted internally, in the future we may want to support fetching +# runbooks from external sources as well. +class RunbookFetcher(Tool): + toolset: "RunbookToolset" + + def __init__(self, toolset: "RunbookToolset"): + super().__init__( + name="fetch_runbook", + description="Get runbook content by runbook link. Use this to get troubleshooting steps for incidents", + parameters={ + # use link as a more generic term for runbook path, considering we may have external links in the future + "link": ToolParameter( + description="The link to the runbook", + type="string", + required=True, + ), + }, + toolset=toolset, # type: ignore + ) + + def _invoke(self, params: Any) -> StructuredToolResult: + path: str = params["link"] + + runbook_path = get_runbook_by_path(path) + try: + with open(runbook_path, "r") as file: + content = file.read() + return StructuredToolResult( + status=ToolResultStatus.SUCCESS, + data=content, + params=params, + ) + except Exception as e: + err_msg = f"Failed to read runbook {runbook_path}: {str(e)}" + logging.error(err_msg) + return StructuredToolResult( + status=ToolResultStatus.ERROR, + error=err_msg, + params=params, + ) + + def get_parameterized_one_liner(self, params) -> str: + path: str = params["link"] + return f"fetched runbook {path}" + + +class RunbookToolset(Toolset): + def __init__(self): + super().__init__( + name="runbook", + description="Fetch runbooks", + icon_url="https://platform.robusta.dev/demos/runbook.svg", + tools=[ + RunbookFetcher(self), + ], + docs_url="https://docs.robusta.dev/master/configuration/holmesgpt/toolsets/runbook.html", + tags=[ + ToolsetTag.CORE, + ], + is_default=True, + ) + + def get_example_config(self) -> Dict[str, Any]: + return {} diff --git a/tests/plugins/prompt/test_generic_ask_conversation.py b/tests/plugins/prompt/test_generic_ask_conversation.py index 863975b158..879deb3fbf 100644 --- a/tests/plugins/prompt/test_generic_ask_conversation.py +++ b/tests/plugins/prompt/test_generic_ask_conversation.py @@ -1,11 +1,12 @@ from holmes.core.tools import ToolsetStatusEnum from holmes.plugins.prompts import load_and_render_prompt +from holmes.plugins.runbooks import load_runbook_catalog from holmes.plugins.toolsets.prometheus.prometheus import PrometheusToolset -template = "builtin://generic_ask_conversation.jinja2" - def test_prometheus_prompt_inclusion(): + template = "builtin://generic_ask_conversation.jinja2" + # Case 1: prometheus/metrics is enabled ts = PrometheusToolset() ts.status = ToolsetStatusEnum.ENABLED @@ -31,3 +32,10 @@ def test_prometheus_prompt_inclusion(): # Check prometheus section is not included assert "# Prometheus/PromQL queries" not in rendered assert "Use prometheus to execute promql queries" not in rendered + + +def test_runbook_prompt(): + template = "builtin://generic_ask.jinja2" + context = {"runbooks": load_runbook_catalog()} + rendered = load_and_render_prompt(template, context) + assert "# Available Runbooks" in rendered diff --git a/tests/plugins/runbooks/test_catalog.py b/tests/plugins/runbooks/test_catalog.py new file mode 100644 index 0000000000..d532c1d7fa --- /dev/null +++ b/tests/plugins/runbooks/test_catalog.py @@ -0,0 +1,17 @@ +import os + +from holmes.plugins.runbooks import get_runbook_by_path, load_runbook_catalog + + +def test_load_runbook_catalog(): + runbooks = load_runbook_catalog() + assert runbooks is not None + assert len(runbooks.catalog) > 0 + for runbook in runbooks.catalog: + assert runbook.description is not None + assert runbook.link is not None + runbook_link = get_runbook_by_path(runbook.link) + # assert file path exists + assert os.path.exists( + runbook_link + ), f"Runbook link {runbook.link} does not exist at {runbook_link}" diff --git a/tests/plugins/toolsets/test_runbook.py b/tests/plugins/toolsets/test_runbook.py new file mode 100644 index 0000000000..11f7e89162 --- /dev/null +++ b/tests/plugins/toolsets/test_runbook.py @@ -0,0 +1,26 @@ +from holmes.core.tools import ToolResultStatus +from holmes.plugins.toolsets.runbook.runbook_fetcher import ( + RunbookFetcher, + RunbookToolset, +) + + +def test_RunbookFetcher(): + runbook_fetch_tool = RunbookFetcher(RunbookToolset()) + result = runbook_fetch_tool._invoke({"link": "wrong_runbook_path"}) + assert result.status == ToolResultStatus.ERROR + assert result.error is not None + + result = runbook_fetch_tool._invoke( + {"link": "networking/dns_troubleshooting_instructions.md"} + ) + + assert result.status == ToolResultStatus.SUCCESS + assert result.error is None + assert result.data is not None + assert ( + runbook_fetch_tool.get_parameterized_one_liner( + {"link": "networking/dns_troubleshooting_instructions.md"} + ) + == "fetched runbook networking/dns_troubleshooting_instructions.md" + ) diff --git a/tests/plugins/toolsets/test_tool_kafka.py b/tests/plugins/toolsets/test_tool_kafka.py index 3bc5ba035e..66199c6255 100644 --- a/tests/plugins/toolsets/test_tool_kafka.py +++ b/tests/plugins/toolsets/test_tool_kafka.py @@ -87,7 +87,6 @@ def test_topic(admin_client): def test_list_kafka_consumers(kafka_toolset): tool = ListKafkaConsumers(kafka_toolset) result = tool.invoke({}) - print(result) assert "consumer_groups:" in result assert ( tool.get_parameterized_one_liner({}) @@ -136,7 +135,6 @@ def test_describe_topic_with_configuration(kafka_toolset, test_topic): tool = DescribeTopic(kafka_toolset) result = tool.invoke({"topic_name": test_topic, "fetch_configuration": True}) - print(result) assert "configuration:" in result assert "partitions:" in result assert "topic:" in result @@ -162,6 +160,5 @@ def test_tool_error_handling(kafka_toolset): tool = DescribeTopic(kafka_toolset) result = tool.invoke({"topic_name": "non_existent_topic"}) - print(result) assert isinstance(result, str) assert "topic: non_existent_topic" in result