Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
17 changes: 15 additions & 2 deletions holmes/config.py
Original file line number Diff line number Diff line change
Expand Up @@ -13,7 +13,7 @@
from holmes.clients.robusta_client import HolmesInfo, fetch_holmes_info
from holmes.common.env_vars import ROBUSTA_AI, ROBUSTA_API_ENDPOINT, ROBUSTA_CONFIG_PATH
from holmes.core.llm import LLM, DefaultLLM
from holmes.core.runbooks import RunbookManager
from holmes.core.runbooks import RunbookCatalogManager, RunbookManager
from holmes.core.supabase_dal import SupabaseDal
from holmes.core.tool_calling_llm import IssueInvestigator, ToolCallingLLM, ToolExecutor
from holmes.core.toolset_manager import ToolsetManager
Expand Down Expand Up @@ -260,7 +260,20 @@ def create_console_toolcalling_llm(
self, dal: Optional[SupabaseDal] = None
) -> ToolCallingLLM:
tool_executor = self.create_console_tool_executor(dal)
return ToolCallingLLM(tool_executor, self.max_steps, self._get_llm())
all_runbooks: list[str] = []
for runbook_path in self.custom_runbooks:
with open(runbook_path, "r") as file:
runbook = file.read()
all_runbooks.append(runbook)
runbook_catalog_manager = RunbookCatalogManager(
llm=self._get_llm(), runbooks=all_runbooks
)
return ToolCallingLLM(
tool_executor,
self.max_steps,
self._get_llm(),
runbook_catalog_manager=runbook_catalog_manager,
)

def create_toolcalling_llm(
self, dal: Optional[SupabaseDal] = None, model: Optional[str] = None
Expand Down
81 changes: 79 additions & 2 deletions holmes/core/runbooks.py
Original file line number Diff line number Diff line change
@@ -1,6 +1,10 @@
from typing import List
import logging
import os
from typing import List, Optional, Tuple

from holmes.core.issue import Issue
from holmes.plugins.runbooks import Runbook
from holmes.core.llm import LLM
from holmes.plugins.runbooks import Runbook, get_runbook_folder, load_catalog


# TODO: our default prompt has a lot of kubernetes specific stuff - see if we can get that into the runbook
Expand All @@ -24,3 +28,76 @@ def get_instructions_for_issue(self, issue: Issue) -> List[str]:
instructions.append(runbook.instructions)

return instructions


class RunbookCatalogManager:
def __init__(self, llm: LLM, runbooks: Optional[List[str]] = None):
"""
Initialize the RunbookCatalogManager with a list of runbooks and an LLM instance.
:param llm: An instance of LLM to use for generating responses.
:param runbooks: A list of custom runbooks. The custom runbooks will be returned without using the LLM.
"""

self.runbooks = runbooks
self.ai = llm
self.catalog = load_catalog()

def get_runbook_by_question(
self, question: str
) -> Tuple[Optional[str], Optional[str]]:
"""
Get the runbook content from the catalog based on the user question by LLM.
If the runbook is not found or is empty, return None.
"""

# TODO(mainred): currently we simply combine all runbooks into a single string and return it,
# but we can consider selecting a specific runbook based on the question.
if self.runbooks:
combined_runbooks = ""
for runbook_str in self.runbooks:
combined_runbooks += f"* {runbook_str}\n"
logging.debug("Custom runbooks are returned.")
return combined_runbooks, None

if not self.catalog:
logging.debug("Runbook catalog is not loaded.")
return None, None

messages = [
{
"role": "system",
"content": f"""
You are an assistant helping the user get the correct runbook to investigate the user question.
You are provided with a catalog of available runbooks with each entry including description and link, and you should return the link field when the description matches the user question.
When no runbook matches the user question, you should return empty string.
Here is the catalog of available runbooks:
{self.catalog.model_dump_json()}.
""",
},
{"role": "user", "content": f"{question}"},
]
response = self.ai.completion(messages, temperature=0)
# when no runbook matches the user question, llm returns ""
runbook_abs_link = response.choices[0].message.content.strip(' "') # type: ignore
if len(runbook_abs_link) == 0:
logging.debug("No runbook found for the question.")
return None, None
else:
logging.debug(f"Runbook link from LLM: {runbook_abs_link}")
runbook_folder = get_runbook_folder()

runbookPath = os.path.join(runbook_folder, runbook_abs_link)
try:
with open(runbookPath, "r") as file:
content = file.read()
if len(content.strip()) == 0:
logging.warning(f"The runbook '{runbookPath}' is empty.")
return None, runbook_abs_link
# If the file is found and not empty, return its content
return content, runbook_abs_link
Comment thread
mainred marked this conversation as resolved.
except FileNotFoundError:
logging.error(f"The file '{runbookPath}' was not found.")
return None, None
except Exception as e:
logging.error(f"An error occurred while reading the file: {e}")
return None, None
36 changes: 34 additions & 2 deletions holmes/core/tool_calling_llm.py
Original file line number Diff line number Diff line change
Expand Up @@ -29,7 +29,7 @@
from holmes.core.llm import LLM
from holmes.core.performance_timing import PerformanceTiming
from holmes.core.resource_instruction import ResourceInstructions
from holmes.core.runbooks import RunbookManager
from holmes.core.runbooks import RunbookCatalogManager, RunbookManager
from holmes.core.tools import StructuredToolResult, ToolExecutor, ToolResultStatus
from holmes.plugins.prompts import load_and_render_prompt
from holmes.utils.global_instructions import (
Expand Down Expand Up @@ -115,10 +115,17 @@ def get_tool_usage_summary(self):
class ToolCallingLLM:
llm: LLM

def __init__(self, tool_executor: ToolExecutor, max_steps: int, llm: LLM):
def __init__(
self,
tool_executor: ToolExecutor,
max_steps: int,
llm: LLM,
runbook_catalog_manager: Optional[RunbookCatalogManager] = None,
):
self.tool_executor = tool_executor
self.max_steps = max_steps
self.llm = llm
self.runbook_catalog_manager = runbook_catalog_manager

def prompt_call(
self,
Expand Down Expand Up @@ -161,6 +168,27 @@ def call( # type: ignore
tool_calls = [] # type: ignore
tools = self.tool_executor.get_all_tools_openai_format()
perf_timing.measure("get_all_tools_openai_format")

# append user questions with runbook selected from runbook catalog
# Normally user_question should exists
user_prompt = ""
for message in messages:
if message.get("role") == "user":
user_prompt = message.get("content", "")
if self.runbook_catalog_manager:
runbook, link = self.runbook_catalog_manager.get_runbook_by_question(
user_prompt
)
if runbook:
logging.info(f"Found runbook {link} for question '{user_prompt}'")
user_prompt_with_runbook = add_runbook_to_user_prompt(
user_prompt, runbook
) # type: ignore
for msg in messages:
if msg.get("role") == "user":
msg["content"] = user_prompt_with_runbook
break

max_steps = self.max_steps
i = 0

Expand Down Expand Up @@ -718,3 +746,7 @@ def investigate(

def create_sse_message(event_type: str, data: dict = {}):
return f"event: {event_type}\ndata: {json.dumps(data)}\n\n"


def add_runbook_to_user_prompt(user_prompt: Optional[str], runbook: str) -> str:
return f"My instructions to check '{user_prompt}' by following the runbook:\n {runbook}"
6 changes: 5 additions & 1 deletion holmes/main.py
Original file line number Diff line number Diff line change
@@ -1,10 +1,10 @@
# ruff: noqa: E402
import os
from holmes.core.prompt import append_file_to_user_prompt
from collections import OrderedDict

from tabulate import tabulate # type: ignore

from holmes.core.prompt import append_file_to_user_prompt
from holmes.utils.cert_utils import add_custom_certificate

ADDITIONAL_CERTIFICATE: str = os.environ.get("CERTIFICATE", "")
Expand Down Expand Up @@ -333,6 +333,7 @@ def ask(
model: Optional[str] = opt_model,
config_file: Optional[Path] = opt_config_file,
custom_toolsets: Optional[List[Path]] = opt_custom_toolsets,
custom_runbooks: Optional[List[Path]] = opt_custom_runbooks,
max_steps: Optional[int] = opt_max_steps,
verbose: Optional[List[bool]] = opt_verbose,
# semi-common options
Expand Down Expand Up @@ -386,11 +387,13 @@ def ask(
custom_toolsets_from_cli=custom_toolsets,
slack_token=slack_token,
slack_channel=slack_channel,
custom_runbooks=custom_runbooks,
)

ai = config.create_console_toolcalling_llm(
dal=None, # type: ignore
)

template_context = {
"toolsets": ai.tool_executor.toolsets,
}
Expand All @@ -414,6 +417,7 @@ def ask(

if echo_request:
console.print("[bold yellow]User:[/bold yellow] " + initial_user_prompt)

for path in include_file: # type: ignore
initial_user_prompt = append_file_to_user_prompt(initial_user_prompt, path)
console.print(f"[bold yellow]Loading file {path}[/bold yellow]")
Expand Down
22 changes: 22 additions & 0 deletions holmes/plugins/runbooks/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,22 @@
# Runbooks

Runbooks folder contains operational runbooks for the HolmesGPT project. Runbooks provide step-by-step instructions for common tasks, troubleshooting, and maintenance procedures related to the plugins in this directory.

## Purpose

- Standardize operational processes
- Enable quick onboarding for new team members
- Reduce downtime by providing clear troubleshooting steps

## Structure

### Structured Runbook

Structured runbooks are designed for specific issues when conditions like issue name, id or source match, the corresponding instructions will be returned for investigation.
For example, the investigation in [kube-prometheus-stack.yaml](kube-prometheus-stack.yaml) will be returned when the issue to be investigated match either KubeSchedulerDown or KubeControllerManagerDown.
This runbook is mainly used for `holmes investigate`

### Catalog

Catalog specified in [catalog.json](catalog.json) contains a collection of runbooks written in markdown.
During runtime, LLM will compare the runbook description with the user question and return the most matched runbook for investigation. It's possible no runbook is returned for no match.
42 changes: 42 additions & 0 deletions holmes/plugins/runbooks/__init__.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,7 @@
import json
import os
import os.path
from datetime import date
from pathlib import Path
from typing import List, Optional, Pattern, Union

Expand All @@ -9,6 +11,8 @@

THIS_DIR = os.path.abspath(os.path.dirname(__file__))

CATALOG_FILE = "catalog.json"


class IssueMatcher(RobustaBaseConfig):
issue_id: Optional[Pattern] = None # unique id
Expand Down Expand Up @@ -48,3 +52,41 @@ def load_builtin_runbooks() -> List[Runbook]:
path = os.path.join(THIS_DIR, filename)
all_runbooks.extend(load_runbooks_from_file(path))
return all_runbooks


class RunbookCatalogEntry(BaseModel):
"""
RunbookCatalogEntry contains metadata about a runbook
Different from runbooks provided by Runbook class, this entry points to markdown file containing the runbook content.
"""

Update_Date: date
Description: str
KeyWords: list[str]
link: str


class RunbookCatalog(BaseModel):
"""
RunbookCatalog is a collection of runbook entries, each entry contains metadata about the runbook.
The correct runbook can be selected from the list by comparing the description with the user question.
"""

catalog: List[RunbookCatalogEntry]


def load_catalog() -> Optional[RunbookCatalog]:
dir_path = os.path.dirname(os.path.realpath(__file__))

catalogPath = os.path.join(dir_path, CATALOG_FILE)
if not os.path.isfile(catalogPath):
return None

with open(catalogPath) as file:
catalog_dict = json.load(file)
return RunbookCatalog(**catalog_dict)
return None


def get_runbook_folder() -> str:
return os.path.dirname(os.path.realpath(__file__))
14 changes: 14 additions & 0 deletions holmes/plugins/runbooks/catalog.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,14 @@
{
"catalog": [
{
"Update_Date": "2025-06-17",
"Description": "Runbook to investigate DNS resolution issue on Kubernetes cluster",
"KeyWords": [
"networking",
"CoreDNS",
"DNS"
],
"link": "networking/dns_troubleshooting_instructions.md"
}
]
}
Original file line number Diff line number Diff line change
@@ -0,0 +1,66 @@
# DNS Troubleshooting Guidelines (Kubernetes)

## Goal
Your primary goal when using these tools is to diagnose DNS resolution issues within a Kubernetes cluster, focusing on identifying common problems like incorrect CoreDNS/kube-dns setup, network policies, or service discovery failures by strictly following the workflow for DNS diagnosis.

* Use the tools to gather information about the DNS pods, services, and configuration.
* Clearly present the key findings from the tool outputs in your analysis.
* Instead of provide next steps to the user, you need to follow the troubleshoot guide to execute the steps.
* When getting pod logs, always try to get the log filter by log_filter toolset to filter out unnecessary logs by tool kubectl_logs_grep_no_match

## Workflow for DNS Diagnosis

1. **Check CoreDNS/kube-dns Pods:**
* Verify that the DNS pods (e.g., CoreDNS or kube-dns) are running in the `kube-system` namespace.
* Look for restarts or crashes in the DNS pods.

2. **Examine DNS Service:**
* Ensure the DNS service is correctly defined: `kubectl get svc kube-dns -n kube-system` (or the equivalent for your DNS provider).
* Verify the ClusterIP of the DNS service and the ports (usually 53/UDP and 53/TCP).

3. **Test DNS Resolution from a Pod:**
* Launch a debugging pod (e.g., using `busybox` or `nslookup` tools).
* **Inside the debug pod:**
* Check `/etc/resolv.conf`:
* The `nameserver` should point to the DNS service's ClusterIP.
* The `search` path should be appropriate for your namespaces (e.g., `your-namespace.svc.cluster.local svc.cluster.local cluster.local`).
* The `options` (like `ndots:5`) can affect resolution behavior.
* Attempt to resolve internal cluster names:
* A service in the same namespace (e.g., `myservice`).
* A service in a different namespace (e.g., `myservice.othernamespace`).
* A fully qualified domain name (FQDN) (e.g., `myservice.othernamespace.svc.cluster.local`).
* Attempt to resolve external names (e.g., `www.google.com`).
* Use tools like `nslookup <hostname>` or `dig <hostname>` for detailed query information.

4. **Check NetworkPolicies:**
* If NetworkPolicies are in place, ensure they allow DNS traffic (to port 53 UDP/TCP) from your application pods to the DNS pods/service.
* List NetworkPolicies and Examine policies that might be affecting the source or destination pods.

5. **Review CoreDNS Configuration (if applicable):**
* Inspect the CoreDNS ConfigMap: `kubectl get configmap coredns -n kube-system -o yaml`.
* Look for errors or misconfigurations in the Corefile (e.g., incorrect upstream resolvers, plugin issues).
* Inspect the customized CoreDNS ConfigMap: `kubectl get configmap coredns-custom -n kube-system -o yaml`.
* Look for errors or misconfigurations in the customizated CoreDNS config (e.g., incorrect upstream resolvers, plugin issues).

6. **Check the DNS trace**
* Use findings from the DNS trace to pinpoint where DNS resolution is failing (e.g., query not reaching DNS server, invalid FQDN, or error response from DNS server).
* DNS Server should always respond to the requests from the client. Valid FQDN should return NOERROR, and invalid FQDN should return NXDOMAIN

## Synthesize Findings
Based on the outputs from the above steps, describe the DNS issue clearly. For example:
* "DNS resolution for internal service 'myservice' is failing from pods in namespace 'app-ns'. The CoreDNS pods in `kube-system` are running but show 'connection refused' errors in their logs when trying to reach upstream resolvers."
* "Pods in namespace 'secure-ns' cannot resolve any hostnames. `/etc/resolv.conf` in these pods is missing the correct `nameserver` entry. This is likely due to a misconfiguration in the pod's `dnsPolicy` or the underlying node's DNS setup."
* "External DNS resolution is failing cluster-wide. The CoreDNS ConfigMap shows that the `forward` plugin is pointing to an incorrect upstream DNS server IP address."
* "DNS lookups for 'service-a.namespace-b' are timing out. A NetworkPolicy in 'namespace-b' is blocking egress traffic on port 53 to the kube-dns service."

## Recommend Remediation Steps (Based on Docs)
* **CRITICAL:** ALWAYS refer to the official Kubernetes DNS debugging guide for detailed troubleshooting and solutions:
* Main guide: https://kubernetes.io/docs/tasks/administer-cluster/dns-debugging-resolution/
* CoreDNS specific: https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/ (for CoreDNS customization which might be relevant)
* **DO NOT invent recovery procedures.** Your role is to diagnose and *point* to the correct documentation or standard procedures.
* Based on the findings, suggest which sections of the documentation are most relevant.
* If DNS pods are not running, guide towards checking pod deployment and node health.
* If `/etc/resolv.conf` is incorrect, point to sections on Pod `dnsPolicy` and `dnsConfig`.
* If NetworkPolicies are suspected, suggest reviewing policy definitions to allow DNS.
* If CoreDNS configuration seems problematic, refer to CoreDNS documentation and the Kubernetes guide on customizing it.
* If upstream DNS resolution is failing, suggest checking the upstream DNS servers and CoreDNS forward configuration.