Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
16 changes: 15 additions & 1 deletion holmes/config.py
Original file line number Diff line number Diff line change
Expand Up @@ -18,7 +18,11 @@
from holmes.core.tool_calling_llm import IssueInvestigator, ToolCallingLLM, ToolExecutor
from holmes.core.toolset_manager import ToolsetManager
from holmes.plugins.destinations.slack import SlackDestination
from holmes.plugins.runbooks import load_builtin_runbooks, load_runbooks_from_file
from holmes.plugins.runbooks import (
load_builtin_runbooks,
load_runbook_catalog,
load_runbooks_from_file,
)
from holmes.plugins.sources.github import GitHubSource
from holmes.plugins.sources.jira import JiraServiceManagementSource, JiraSource
from holmes.plugins.sources.opsgenie import OpsGenieSource
Expand Down Expand Up @@ -230,6 +234,16 @@ def __get_cluster_name() -> Optional[str]:

return None

@staticmethod
def get_runbook_catalog() -> str:
# TODO(mainred): besides the built-in runbooks, we need to allow the user to bring their own runbooks
runbook_catalog = load_runbook_catalog()
if runbook_catalog is not None:
Comment thread
arikalon1 marked this conversation as resolved.
return runbook_catalog.model_dump_json()
else:
logging.warning("Runbook catalog not found")
return json.dumps({"catalog": []})

def create_console_tool_executor(self, dal: Optional[SupabaseDal]) -> ToolExecutor:
"""
Creates a ToolExecutor instance configured for CLI usage. This executor manages the available tools
Expand Down
4 changes: 2 additions & 2 deletions holmes/main.py
Original file line number Diff line number Diff line change
Expand Up @@ -26,7 +26,6 @@
from rich.markdown import Markdown
from rich.rule import Rule


from holmes import get_version # type: ignore
from holmes.config import (
DEFAULT_CONFIG_LOCATION,
Expand All @@ -38,12 +37,12 @@
from holmes.core.resource_instruction import ResourceInstructionDocument
from holmes.core.tool_calling_llm import LLMResult
from holmes.core.tools import pretty_print_toolset_status
from holmes.interactive import run_interactive_loop
from holmes.plugins.destinations import DestinationType
from holmes.plugins.interfaces import Issue
from holmes.plugins.prompts import load_and_render_prompt
from holmes.plugins.sources.opsgenie import OPSGENIE_TEAM_INTEGRATION_KEY_HELP
from holmes.utils.file_utils import write_json_file
from holmes.interactive import run_interactive_loop

app = typer.Typer(add_completion=False, pretty_exceptions_show_locals=False)
investigate_app = typer.Typer(
Expand Down Expand Up @@ -320,6 +319,7 @@ def ask(
)
template_context = {
"toolsets": ai.tool_executor.toolsets,
"runbooks": config.get_runbook_catalog(),
}

system_prompt_rendered = load_and_render_prompt(system_prompt, template_context) # type: ignore
Expand Down
2 changes: 1 addition & 1 deletion holmes/plugins/prompts/_general_instructions.jinja2
Original file line number Diff line number Diff line change
Expand Up @@ -75,4 +75,4 @@ Reminder:
* That is different than - for example - fetching a pod's logs and seeing that the pod itself has permission errors. in that case, you explain say that permission errors are the cause of the problem and give details
* Issues are a subset of findings. When asked about an issue or a finding and you have an id, use the tool `fetch_finding_by_id`.
* For any question, try to make the answer specific to the user's cluster.
** For example, if asked to port forward, find out the app or pod port (kubectl decribe) and provide a port forward command specific to the user's question
** For example, if asked to port forward, find out the app or pod port (kubectl describe) and provide a port forward command specific to the user's question
17 changes: 17 additions & 0 deletions holmes/plugins/prompts/_runbook_instructions.jinja2
Original file line number Diff line number Diff line change
@@ -0,0 +1,17 @@
{% if runbooks and runbooks.catalog|length > 0 %}
# Runbook Selection

# Available Runbooks

{%- for runbook in runbooks.catalog -%}

description: {{ runbook.description }}
link: {{ runbook.link }}

{%- endfor -%}

ALWAYS try to find the runbooks that can provide troubleshooting instructions when the user describes an operational issue, debugging scenario, or asks for step‑by‑step troubleshooting.
To get the runbook details, use `fetch_runbook` tool by comparing the runbook description with the user prompt.
ALWAYS follow the steps described in the runbook.
If you decided not to follow one or more steps, ALWAYS explain why.
{%- endif -%}
2 changes: 2 additions & 0 deletions holmes/plugins/prompts/generic_ask.jinja2
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,8 @@ Bias towards not asking the user for help if you can find the answer yourself.

{% include '_general_instructions.jinja2' %}

{% include '_runbook_instructions.jinja2' %}

# Style guide

* Reply with terse output.
Expand Down
22 changes: 22 additions & 0 deletions holmes/plugins/runbooks/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,22 @@
# Runbooks

Runbooks folder contains operational runbooks for the HolmesGPT project. Runbooks provide step-by-step instructions for common tasks, troubleshooting, and maintenance procedures related to the plugins in this directory.

## Purpose

- Standardize operational processes
- Enable quick onboarding for new team members
- Reduce downtime by providing clear troubleshooting steps

## Structure

### Structured Runbook

Structured runbooks are designed for specific issues when conditions like issue name, id or source match, the corresponding instructions will be returned for investigation.
For example, the investigation in [kube-prometheus-stack.yaml](kube-prometheus-stack.yaml) will be returned when the issue to be investigated match either KubeSchedulerDown or KubeControllerManagerDown.
This runbook is mainly used for `holmes investigate`

### Catalog

Catalog specified in [catalog.json](catalog.json) contains a collection of runbooks written in markdown.
During runtime, LLM will compare the runbook description with the user question and return the most matched runbook for investigation. It's possible no runbook is returned for no match.
50 changes: 50 additions & 0 deletions holmes/plugins/runbooks/__init__.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,8 @@
import json
import logging
import os
import os.path
from datetime import date
from pathlib import Path
from typing import List, Optional, Pattern, Union

Expand All @@ -9,6 +12,8 @@

THIS_DIR = os.path.abspath(os.path.dirname(__file__))

CATALOG_FILE = "catalog.json"


class IssueMatcher(RobustaBaseConfig):
issue_id: Optional[Pattern] = None # unique id
Expand Down Expand Up @@ -48,3 +53,48 @@ def load_builtin_runbooks() -> List[Runbook]:
path = os.path.join(THIS_DIR, filename)
all_runbooks.extend(load_runbooks_from_file(path))
return all_runbooks


class RunbookCatalogEntry(BaseModel):
"""
RunbookCatalogEntry contains metadata about a runbook
Different from runbooks provided by Runbook class, this entry points to markdown file containing the runbook content.
"""

update_date: date
description: str
link: str


class RunbookCatalog(BaseModel):
"""
RunbookCatalog is a collection of runbook entries, each entry contains metadata about the runbook.
The correct runbook can be selected from the list by comparing the description with the user question.
"""

catalog: List[RunbookCatalogEntry]


def load_runbook_catalog() -> Optional[RunbookCatalog]:
dir_path = os.path.dirname(os.path.realpath(__file__))

catalogPath = os.path.join(dir_path, CATALOG_FILE)
if not os.path.isfile(catalogPath):
return None
try:
with open(catalogPath) as file:
catalog_dict = json.load(file)
return RunbookCatalog(**catalog_dict)
except json.JSONDecodeError as e:
logging.error(f"Error decoding JSON from {catalogPath}: {e}")
except Exception as e:
logging.error(
f"Unexpected error while loading runbook catalog from {catalogPath}: {e}"
)
return None


def get_runbook_by_path(runbook_relative_path: str) -> str:
runbook_folder = os.path.dirname(os.path.realpath(__file__))
runbook_path = os.path.join(runbook_folder, runbook_relative_path)
return runbook_path
Comment thread
arikalon1 marked this conversation as resolved.
9 changes: 9 additions & 0 deletions holmes/plugins/runbooks/catalog.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,9 @@
{
"catalog": [
{
"update_date": "2025-06-17",
"description": "Runbook to investigate DNS resolution issue on Kubernetes cluster",
"link": "networking/dns_troubleshooting_instructions.md"
}
]
}
Original file line number Diff line number Diff line change
@@ -0,0 +1,66 @@
# DNS Troubleshooting Guidelines (Kubernetes)

## Goal
Your primary goal when using these tools is to diagnose DNS resolution issues within a Kubernetes cluster, focusing on identifying common problems like incorrect CoreDNS/kube-dns setup, network policies, or service discovery failures by strictly following the workflow for DNS diagnosis.

* Use the tools to gather information about the DNS pods, services, and configuration.
* Clearly present the key findings from the tool outputs in your analysis.
* Instead of provide next steps to the user, you need to follow the troubleshoot guide to execute the steps.
* When getting pod logs, always try to get the log filter by log_filter toolset to filter out unnecessary logs by tool kubectl_logs_grep_no_match

## Workflow for DNS Diagnosis

1. **Check CoreDNS/kube-dns Pods:**
* Verify that the DNS pods (e.g., CoreDNS or kube-dns) are running in the `kube-system` namespace.
* Look for restarts or crashes in the DNS pods.

2. **Examine DNS Service:**
* Ensure the DNS service is correctly defined: `kubectl get svc kube-dns -n kube-system` (or the equivalent for your DNS provider).
* Verify the ClusterIP of the DNS service and the ports (usually 53/UDP and 53/TCP).

3. **Test DNS Resolution from a Pod:**
* Launch a debugging pod (e.g., using `busybox` or `nslookup` tools).
* **Inside the debug pod:**
* Check `/etc/resolv.conf`:
* The `nameserver` should point to the DNS service's ClusterIP.
* The `search` path should be appropriate for your namespaces (e.g., `your-namespace.svc.cluster.local svc.cluster.local cluster.local`).
* The `options` (like `ndots:5`) can affect resolution behavior.
* Attempt to resolve internal cluster names:
* A service in the same namespace (e.g., `myservice`).
* A service in a different namespace (e.g., `myservice.othernamespace`).
* A fully qualified domain name (FQDN) (e.g., `myservice.othernamespace.svc.cluster.local`).
* Attempt to resolve external names (e.g., `www.google.com`).
* Use tools like `nslookup <hostname>` or `dig <hostname>` for detailed query information.

4. **Check NetworkPolicies:**
* If NetworkPolicies are in place, ensure they allow DNS traffic (to port 53 UDP/TCP) from your application pods to the DNS pods/service.
* List NetworkPolicies and Examine policies that might be affecting the source or destination pods.

5. **Review CoreDNS Configuration (if applicable):**
* Inspect the CoreDNS ConfigMap: `kubectl get configmap coredns -n kube-system -o yaml`.
* Look for errors or misconfigurations in the Corefile (e.g., incorrect upstream resolvers, plugin issues).
* Inspect the customized CoreDNS ConfigMap: `kubectl get configmap coredns-custom -n kube-system -o yaml`.
* Look for errors or misconfigurations in the customizated CoreDNS config (e.g., incorrect upstream resolvers, plugin issues).

6. **Check the DNS trace**
* Use findings from the DNS trace to pinpoint where DNS resolution is failing (e.g., query not reaching DNS server, invalid FQDN, or error response from DNS server).
* DNS Server should always respond to the requests from the client. Valid FQDN should return NOERROR, and invalid FQDN should return NXDOMAIN

## Synthesize Findings
Based on the outputs from the above steps, describe the DNS issue clearly. For example:
* "DNS resolution for internal service 'myservice' is failing from pods in namespace 'app-ns'. The CoreDNS pods in `kube-system` are running but show 'connection refused' errors in their logs when trying to reach upstream resolvers."
* "Pods in namespace 'secure-ns' cannot resolve any hostnames. `/etc/resolv.conf` in these pods is missing the correct `nameserver` entry. This is likely due to a misconfiguration in the pod's `dnsPolicy` or the underlying node's DNS setup."
* "External DNS resolution is failing cluster-wide. The CoreDNS ConfigMap shows that the `forward` plugin is pointing to an incorrect upstream DNS server IP address."
* "DNS lookups for 'service-a.namespace-b' are timing out. A NetworkPolicy in 'namespace-b' is blocking egress traffic on port 53 to the kube-dns service."

## Recommend Remediation Steps (Based on Docs)
* **CRITICAL:** ALWAYS refer to the official Kubernetes DNS debugging guide for detailed troubleshooting and solutions:
* Main guide: https://kubernetes.io/docs/tasks/administer-cluster/dns-debugging-resolution/
* CoreDNS specific: https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/ (for CoreDNS customization which might be relevant)
* **DO NOT invent recovery procedures.** Your role is to diagnose and *point* to the correct documentation or standard procedures.
* Based on the findings, suggest which sections of the documentation are most relevant.
* If DNS pods are not running, guide towards checking pod deployment and node health.
* If `/etc/resolv.conf` is incorrect, point to sections on Pod `dnsPolicy` and `dnsConfig`.
* If NetworkPolicies are suspected, suggest reviewing policy definitions to allow DNS.
* If CoreDNS configuration seems problematic, refer to CoreDNS documentation and the Kubernetes guide on customizing it.
* If upstream DNS resolution is failing, suggest checking the upstream DNS servers and CoreDNS forward configuration.
4 changes: 3 additions & 1 deletion holmes/plugins/toolsets/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,13 +9,13 @@
import holmes.utils.env as env_utils
from holmes.core.supabase_dal import SupabaseDal
from holmes.core.tools import Toolset, ToolsetType, ToolsetYamlFromConfig, YAMLToolset
from holmes.plugins.toolsets.bash.bash_toolset import BashExecutorToolset
from holmes.plugins.toolsets.coralogix.toolset_coralogix_logs import (
CoralogixLogsToolset,
)
from holmes.plugins.toolsets.datadog import DatadogToolset
from holmes.plugins.toolsets.git import GitToolset
from holmes.plugins.toolsets.grafana.toolset_grafana import GrafanaToolset
from holmes.plugins.toolsets.bash.bash_toolset import BashExecutorToolset
from holmes.plugins.toolsets.grafana.toolset_grafana_loki import GrafanaLokiToolset
from holmes.plugins.toolsets.grafana.toolset_grafana_tempo import GrafanaTempoToolset
from holmes.plugins.toolsets.internet.internet import InternetToolset
Expand All @@ -29,6 +29,7 @@
from holmes.plugins.toolsets.prometheus.prometheus import PrometheusToolset
from holmes.plugins.toolsets.rabbitmq.toolset_rabbitmq import RabbitMQToolset
from holmes.plugins.toolsets.robusta.robusta import RobustaToolset
from holmes.plugins.toolsets.runbook.runbook_fetcher import RunbookToolset

THIS_DIR = os.path.abspath(os.path.dirname(__file__))

Expand Down Expand Up @@ -70,6 +71,7 @@ def load_python_toolsets(dal: Optional[SupabaseDal]) -> List[Toolset]:
RabbitMQToolset(),
GitToolset(),
BashExecutorToolset(),
RunbookToolset(),
]

return toolsets
Expand Down
Empty file.
78 changes: 78 additions & 0 deletions holmes/plugins/toolsets/runbook/runbook_fetcher.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,78 @@
import logging
from typing import Any, Dict

from holmes.core.tools import (
StructuredToolResult,
Tool,
ToolParameter,
ToolResultStatus,
Toolset,
ToolsetTag,
)
from holmes.plugins.runbooks import get_runbook_by_path


# TODO(mainred): currently we support fetch runbooks hosted internally, in the future we may want to support fetching
# runbooks from external sources as well.
class RunbookFetcher(Tool):
toolset: "RunbookToolset"

def __init__(self, toolset: "RunbookToolset"):
super().__init__(
name="fetch_runbook",
description="Get runbook content by runbook link. Use this to get troubleshooting steps for incidents",
parameters={
# use link as a more generic term for runbook path, considering we may have external links in the future
"link": ToolParameter(
description="The link to the runbook",
type="string",
required=True,
),
},
toolset=toolset, # type: ignore
)

def _invoke(self, params: Any) -> StructuredToolResult:
path: str = params["link"]

runbook_path = get_runbook_by_path(path)
try:
with open(runbook_path, "r") as file:
content = file.read()
return StructuredToolResult(
status=ToolResultStatus.SUCCESS,
data=content,
params=params,
)
except Exception as e:
err_msg = f"Failed to read runbook {runbook_path}: {str(e)}"
logging.error(err_msg)
return StructuredToolResult(
status=ToolResultStatus.ERROR,
error=err_msg,
params=params,
)

def get_parameterized_one_liner(self, params) -> str:
path: str = params["link"]
return f"fetched runbook {path}"


class RunbookToolset(Toolset):
def __init__(self):
super().__init__(
name="runbook",
description="Fetch runbooks",
icon_url="https://platform.robusta.dev/demos/runbook.svg",
tools=[
RunbookFetcher(self),
],
docs_url="https://docs.robusta.dev/master/configuration/holmesgpt/toolsets/runbook.html",
tags=[
ToolsetTag.CORE,
],
is_default=True,
)

def get_example_config(self) -> Dict[str, Any]:
return {}
12 changes: 10 additions & 2 deletions tests/plugins/prompt/test_generic_ask_conversation.py
Original file line number Diff line number Diff line change
@@ -1,11 +1,12 @@
from holmes.core.tools import ToolsetStatusEnum
from holmes.plugins.prompts import load_and_render_prompt
from holmes.plugins.runbooks import load_runbook_catalog
from holmes.plugins.toolsets.prometheus.prometheus import PrometheusToolset

template = "builtin://generic_ask_conversation.jinja2"


def test_prometheus_prompt_inclusion():
template = "builtin://generic_ask_conversation.jinja2"

# Case 1: prometheus/metrics is enabled
ts = PrometheusToolset()
ts.status = ToolsetStatusEnum.ENABLED
Expand All @@ -31,3 +32,10 @@ def test_prometheus_prompt_inclusion():
# Check prometheus section is not included
assert "# Prometheus/PromQL queries" not in rendered
assert "Use prometheus to execute promql queries" not in rendered


def test_runbook_prompt():
template = "builtin://generic_ask.jinja2"
context = {"runbooks": load_runbook_catalog()}
rendered = load_and_render_prompt(template, context)
assert "# Available Runbooks" in rendered
Loading