diff --git a/docs/reference/http-api.md b/docs/reference/http-api.md index d43a4221ab..4ec4f53a0c 100644 --- a/docs/reference/http-api.md +++ b/docs/reference/http-api.md @@ -1,7 +1,7 @@ # HolmesGPT API Reference ## Overview -The HolmesGPT API provides endpoints for automated investigations, workload health checks, and conversational troubleshooting. This document describes each endpoint, its purpose, request fields, and example usage. +The HolmesGPT API provides endpoints for automated investigations and conversational troubleshooting. This document describes each endpoint, its purpose, request fields, and example usage. ## Model Parameter Behavior @@ -245,103 +245,6 @@ curl -X POST http:///api/issue_chat \ --- -### `/api/workload_health_check` (POST) -**Description:** Performs a health check on a specified workload (e.g., a Kubernetes deployment). - -#### Request Fields - -| Field | Required | Default | Type | Description | -|-------------------------|----------|--------------------------------------------|-----------|--------------------------------------------------| -| ask | Yes | | string | User's question | -| resource | Yes | | object | Resource details (e.g., name, kind) | -| alert_history_since_hours| No | 24 | float | How many hours back to check alerts | -| alert_history | No | true | boolean | Whether to include alert history | -| stored_instructions | No | true | boolean | Use stored instructions | -| instructions | No | [] | list | Additional instructions | -| include_tool_calls | No | false | boolean | Include tool calls in response | -| include_tool_call_results| No | false | boolean | Include tool call results in response | -| prompt_template | No | "builtin://kubernetes_workload_ask.jinja2" | string | Prompt template to use | -| model | No | | string | Model name from your `modelList` configuration | - -**Example** -```bash -curl -X POST http:///api/workload_health_check \ - -H "Content-Type: application/json" \ - -d '{ - "ask": "Why is my deployment unhealthy?", - "resource": {"name": "my-deployment", "kind": "Deployment"}, - "alert_history_since_hours": 12 - }' -``` - -**Example** Response -```json -{ - "analysis": "Deployment 'my-deployment' is unhealthy due to repeated CrashLoopBackOff events.", - "sections": null, - "tool_calls": [ - { - "tool_call_id": "2", - "tool_name": "kubectl_get_events", - "description": "Fetch recent events", - "result": {"events": "..."} - } - ], - "instructions": [...] -} -``` - ---- - -### `/api/workload_health_chat` (POST) -**Description:** Conversational interface for discussing the health of a workload. - -#### Request Fields - -| Field | Required | Default | Type | Description | -|-------------------------|----------|---------|-----------|--------------------------------------------------| -| ask | Yes | | string | User's question | -| workload_health_result | Yes | | object | Previous health check result (see below) | -| resource | Yes | | object | Resource details | -| conversation_history | No | | list | Conversation history (first message must be system)| -| model | No | | string | Model name from your `modelList` configuration | - -**workload_health_result** object: -- `analysis` (string, optional): Previous analysis -- `tools` (list, optional): Tools used/results - -**Example** -```bash -curl -X POST http:///api/workload_health_chat \ - -H "Content-Type: application/json" \ - -d '{ - "ask": "Check the workload health.", - "workload_health_result": { - "analysis": "Previous health check: all good.", - "tools": [] - }, - "resource": {"name": "my-deployment", "kind": "Deployment"}, - "conversation_history": [ - {"role": "system", "content": "You are a helpful assistant."} - ] - }' -``` - -**Example** Response -```json -{ - "analysis": "The deployment 'my-deployment' is healthy. No recent issues detected.", - "conversation_history": [ - {"role": "system", "content": "You are a helpful assistant."}, - {"role": "user", "content": "Check the workload health."}, - {"role": "assistant", "content": "The deployment 'my-deployment' is healthy. No recent issues detected."} - ], - "tool_calls": [...] -} -``` - ---- - ### `/api/model` (GET) **Description:** Returns a list of available AI models that can be used for investigations and chat. diff --git a/holmes/core/conversations.py b/holmes/core/conversations.py index 1ec88073e5..f5c91bf40e 100644 --- a/holmes/core/conversations.py +++ b/holmes/core/conversations.py @@ -5,7 +5,6 @@ from holmes.core.models import ( ToolCallConversationResult, IssueChatRequest, - WorkloadHealthChatRequest, ) from holmes.plugins.prompts import load_and_render_prompt from holmes.core.tool_calling_llm import ToolCallingLLM @@ -425,224 +424,3 @@ def build_chat_messages( ) truncate_tool_messages(conversation_history, tool_size) # type: ignore return conversation_history # type: ignore - - -def build_workload_health_chat_messages( - workload_health_chat_request: WorkloadHealthChatRequest, - ai: ToolCallingLLM, - config: Config, - global_instructions: Optional[Instructions] = None, - runbooks: Optional[RunbookCatalog] = None, -): - """ - This function generates a list of messages for workload health conversation and ensures that the message sequence adheres to the model's context window limitations - by truncating tool outputs as necessary before sending to llm. - - We always expect conversation_history to be passed in the openAI format which is supported by litellm and passed back by us. - That's why we assume that first message in the conversation is system message and truncate tools for it. - - System prompt handling: - 1. For new conversations (empty conversation_history): - - Creates a new system prompt using kubernetes_workload_chat.jinja2 template - - Includes workload analysis, tools (if any), and resource information - - If there are tools, calculates appropriate tool size and truncates tool outputs - - 2. For existing conversations: - - Preserves the conversation history - - Updates the first message (system prompt) with recalculated content - - Truncates tool outputs if necessary to fit context window - - Maintains the original conversation flow while ensuring context limits - - Example structure of conversation history: - conversation_history = [ - # System prompt with workload analysis - {"role": "system", "content": "...."}, - # User message asking about workload health - {"role": "user", "content": "What's the current health status of my deployment?"}, - # Assistant initiates a tool call - { - "role": "assistant", - "content": None, - "tool_call": { - "name": "check_workload_metrics", - "arguments": "{\"namespace\": \"default\", \"workload\": \"my-deployment\"}" - } - }, - # Tool/Function response - { - "role": "tool", - "name": "check_workload_metrics", - "content": "{\"cpu_usage\": \"45%\", \"memory_usage\": \"60%\", \"status\": \"Running\"}" - }, - # Assistant's final response to the user - { - "role": "assistant", - "content": "Your deployment is running normally with CPU usage at 45% and memory usage at 60%." - }, - ] - """ - - template_path = "builtin://kubernetes_workload_chat.jinja2" - - conversation_history = workload_health_chat_request.conversation_history - user_prompt = workload_health_chat_request.ask - workload_analysis = workload_health_chat_request.workload_health_result.analysis - tools_for_workload = workload_health_chat_request.workload_health_result.tools - resource = workload_health_chat_request.resource - - if not conversation_history or len(conversation_history) == 0: - runbooks_ctx = generate_runbooks_args( - runbook_catalog=runbooks, - global_instructions=global_instructions, - ) - user_prompt = generate_user_prompt( - user_prompt, - runbooks_ctx, - ) - - number_of_tools_for_workload = len(tools_for_workload) # type: ignore - if number_of_tools_for_workload == 0: - system_prompt = load_and_render_prompt( - template_path, - { - "workload_analysis": workload_analysis, - "tools_called_for_workload": tools_for_workload, - "resource": resource, - "toolsets": ai.tool_executor.toolsets, - "cluster_name": config.cluster_name, - "runbooks_enabled": True if runbooks else False, - }, - ) - messages = [ - { - "role": "system", - "content": system_prompt, - }, - { - "role": "user", - "content": user_prompt, - }, - ] - return messages - - template_context_without_tools = { - "workload_analysis": workload_analysis, - "tools_called_for_workload": None, - "resource": resource, - "toolsets": ai.tool_executor.toolsets, - "cluster_name": config.cluster_name, - "runbooks_enabled": True if runbooks else False, - } - system_prompt_without_tools = load_and_render_prompt( - template_path, template_context_without_tools - ) - messages_without_tools = [ - { - "role": "system", - "content": system_prompt_without_tools, - }, - { - "role": "user", - "content": user_prompt, - }, - ] - tool_size = calculate_tool_size( - ai, messages_without_tools, number_of_tools_for_workload - ) - - truncated_workload_result_tool_calls = [ - ToolCallConversationResult( - name=tool.name, - description=tool.description, - output=tool.output[:tool_size], - ) - for tool in tools_for_workload # type: ignore - ] - - truncated_template_context = { - "workload_analysis": workload_analysis, - "tools_called_for_workload": truncated_workload_result_tool_calls, - "resource": resource, - "toolsets": ai.tool_executor.toolsets, - "cluster_name": config.cluster_name, - "runbooks_enabled": True if runbooks else False, - } - system_prompt_with_truncated_tools = load_and_render_prompt( - template_path, truncated_template_context - ) - return [ - { - "role": "system", - "content": system_prompt_with_truncated_tools, - }, - { - "role": "user", - "content": user_prompt, - }, - ] - - runbooks_ctx = generate_runbooks_args( - runbook_catalog=runbooks, - global_instructions=global_instructions, - ) - user_prompt = generate_user_prompt( - user_prompt, - runbooks_ctx, - ) - - conversation_history.append( - { - "role": "user", - "content": user_prompt, - } - ) - number_of_tools = len(tools_for_workload) + len( # type: ignore - [message for message in conversation_history if message.get("role") == "tool"] - ) - - if number_of_tools == 0: - return conversation_history - - conversation_history_without_tools = [ - message for message in conversation_history if message.get("role") != "tool" - ] - template_context_without_tools = { - "workload_analysis": workload_analysis, - "tools_called_for_workload": None, - "resource": resource, - "toolsets": ai.tool_executor.toolsets, - "cluster_name": config.cluster_name, - "runbooks_enabled": True if runbooks else False, - } - system_prompt_without_tools = load_and_render_prompt( - template_path, template_context_without_tools - ) - conversation_history_without_tools[0]["content"] = system_prompt_without_tools - - tool_size = calculate_tool_size( - ai, conversation_history_without_tools, number_of_tools - ) - - truncated_workload_result_tool_calls = [ - ToolCallConversationResult( - name=tool.name, description=tool.description, output=tool.output[:tool_size] - ) - for tool in tools_for_workload # type: ignore - ] - - template_context = { - "workload_analysis": workload_analysis, - "tools_called_for_workload": truncated_workload_result_tool_calls, - "resource": resource, - "toolsets": ai.tool_executor.toolsets, - "cluster_name": config.cluster_name, - "runbooks_enabled": True if runbooks else False, - } - system_prompt_with_truncated_tools = load_and_render_prompt( - template_path, template_context - ) - conversation_history[0]["content"] = system_prompt_with_truncated_tools - - truncate_tool_messages(conversation_history, tool_size) - - return conversation_history diff --git a/holmes/core/models.py b/holmes/core/models.py index 81221f9579..c5a74bd78f 100644 --- a/holmes/core/models.py +++ b/holmes/core/models.py @@ -215,20 +215,6 @@ class IssueChatRequest(ChatRequestBaseModel): investigation_result: IssueInvestigationResult issue_type: str - -class WorkloadHealthRequest(BaseModel): - ask: str - resource: dict - alert_history_since_hours: float = 24 - alert_history: bool = True - stored_instrucitons: bool = True - instructions: Optional[List[str]] = [] - include_tool_calls: bool = False - include_tool_call_results: bool = False - prompt_template: str = "builtin://kubernetes_workload_ask.jinja2" - model: Optional[str] = None - - class ChatRequest(ChatRequestBaseModel): ask: str @@ -247,45 +233,3 @@ class ChatResponse(BaseModel): follow_up_actions: Optional[List[FollowUpAction]] = [] pending_approvals: Optional[List[PendingToolApproval]] = None metadata: Optional[Dict[Any, Any]] = None - - -class WorkloadHealthInvestigationResult(BaseModel): - analysis: Optional[str] = None - tools: Optional[List[ToolCallConversationResult]] = [] - - @model_validator(mode="before") - def check_analysis_and_result(cls, values): - if "result" in values and "analysis" not in values: - values["analysis"] = values["result"] - del values["result"] - return values - - -class WorkloadHealthChatRequest(ChatRequestBaseModel): - ask: str - workload_health_result: WorkloadHealthInvestigationResult - resource: dict - - -workload_health_structured_output = { - "type": "json_schema", - "json_schema": { - "name": "WorkloadHealthResult", - "strict": False, - "schema": { - "type": "object", - "properties": { - "workload_healthy": { - "type": "boolean", - "description": "is the workload in healthy state or in error state", - }, - "root_cause_summary": { - "type": "string", - "description": "concise short explaination leading to the workload_healthy result, pinpoint reason and root cause for the workload issues if any.", - }, - }, - "required": ["root_cause_summary", "workload_healthy"], - "additionalProperties": False, - }, - }, -} diff --git a/holmes/core/supabase_dal.py b/holmes/core/supabase_dal.py index d1c63a98b3..e3df167986 100644 --- a/holmes/core/supabase_dal.py +++ b/holmes/core/supabase_dal.py @@ -666,56 +666,6 @@ def get_ai_credentials(self) -> Tuple[str, str]: return self.account_id, session_token - def get_workload_issues(self, resource: dict, since_hours: float) -> List[str]: - if not self.enabled or not resource: - return [] - - cluster = resource.get("cluster") - if not cluster: - logging.debug("Missing workload cluster for issues.") - return [] - - since: str = (datetime.now() - timedelta(hours=since_hours)).isoformat() - - svc_key = f"{resource.get('namespace', '')}/{resource.get('kind', '')}/{resource.get('name', '')}" - logging.debug(f"getting issues for workload {svc_key}") - try: - res = ( - self.client.table(ISSUES_TABLE) - .select("id, creation_date, aggregation_key") - .eq("account_id", self.account_id) - .eq("cluster", cluster) - .eq("service_key", svc_key) - .gte("creation_date", since) - .order("creation_date") - .execute() - ) - - if not res.data: - return [] - - issue_dict = dict() - for issue in res.data: - issue_dict[issue.get("aggregation_key")] = issue.get("id") - - unique_issues: list[str] = list(issue_dict.values()) - - res = ( - self.client.table(EVIDENCE_TABLE) - .select("data, enrichment_type") - .in_("issue_id", unique_issues) - .not_.in_("enrichment_type", ENRICHMENT_BLACKLIST) - .execute() - ) - - relevant_issues = self.extract_relevant_issues(res) - truncate_evidences_entities_if_necessary(relevant_issues) - return relevant_issues - - except Exception: - logging.exception("failed to fetch workload issues data", exc_info=True) - return [] - def upsert_holmes_status(self, holmes_status_data: dict) -> None: if not self.enabled: logging.info( diff --git a/holmes/plugins/prompts/kubernetes_workload_ask.jinja2 b/holmes/plugins/prompts/kubernetes_workload_ask.jinja2 deleted file mode 100644 index bcc212f456..0000000000 --- a/holmes/plugins/prompts/kubernetes_workload_ask.jinja2 +++ /dev/null @@ -1,77 +0,0 @@ -You are a tool-calling AI assist provided with common devops and IT tools that you can use to troubleshoot problems or answer questions. -Whenever possible you MUST first use tools to investigate then answer the question. -Do not say 'based on the tool output' or explicitly refer to tools at all. -If you output an answer and then realize you need to call more tools or there are possible next steps, you may do so by calling tools at that point in time. - -If the user provides you with extra instructions in a triple single quotes section, ALWAYS perform their instructions and then perform your investigation. - -{% include 'investigation_procedure.jinja2' %} - -{% include '_ai_safety.jinja2' %} - -Global Instructions -You may receive a set of “Global Instructions” that describe how to perform certain tasks, handle certain situations, or apply certain best practices. They are not mandatory for every request, but serve as a reference resource and must be used if the current scenario or user request aligns with one of the described methods or conditions. -Use these rules when deciding how to apply them: - -* Some Global Instructions may describe how to handle specific tasks or scenarios. If the user's current request or the instructions in a triple quotes section reference one of these tasks, ALWAYS follow the Global Instruction for that task. -* Some Global Instructions may define general conditions that always apply if a certain scenario occurs (e.g., "whenever investigating a memory issue, always check resource limits"). If such a condition matches the current situation, apply the Global Instruction accordingly. -* If user's prompt or the instructions in a triple quotes section direct you to perform a task (e.g., “Find owner”) and there is a Global Instruction on how to do that task, ALWAYS follow the Global Instructions on how to perform it. -* If multiple Global Instructions are relevant, apply all that fit. -* If no Global Instruction is relevant, or no condition applies, ignore them and proceed as normal. -* Before finalizing your answer double-check if any Global Instructions apply. If so, ensure you have correctly followed those instructions. - -In general: -* when it can provide extra information, first run as many tools as you need to gather more information, then respond. -* if possible, do so repeatedly with different tool calls each time to gather more information. -* do not stop investigating until you are at the final root cause you are able to find. -* use the "five whys" methodology to find the root cause. -* for example, if you found a problem in microservice A that is due to an error in microservice B, look at microservice B too and find the error in that. -* if you cannot find the resource/application that the user referred to, assume they made a typo or included/excluded characters like - and. -* in this case, try to find substrings or search for the correct spellings -* if you are unable to investigate something properly because you do not have access to the right data, explicitly tell the user that you are missing an integration to access XYZ which you would need to investigate. you should specifically use the templated phrase "I don't have access to
. Please add a Holmes integration for so that I can investigate this." -* always provide detailed information like exact resource names, versions, labels, etc -* even if you found the root cause, keep investigating to find other possible root causes and to gather data for the answer like exact names -* if a runbook url is present as well as tool that can fetch it, you MUST fetch the runbook before beginning your investigation. -* if you don't know, say that the analysis was inconclusive. -* if there are multiple possible causes list them in a numbered list. -* there will often be errors in the data that are not relevant or that do not have an impact - ignore them in your conclusion if you were not able to tie them to an actual error. -* run as many kubectl commands as you need to gather more information, then respond. -* if possible, do so repeatedly on different Kubernetes objects. -* for example, for deployments first run kubectl on the deployment then a replicaset inside it, then a pod inside that. -* when investigating a pod that crashed or application errors, always run kubectl_describe and fetch the pod's logs so that you see current logs and any logs from before a crash. -* do not give an answer like "The pod is pending" as that doesn't state why the pod is pending and how to fix it. -* do not give an answer like "Pod's node affinity/selector doesn't match any available nodes" because that doesn't include data on WHICH label doesn't match -* if investigating an issue on many pods, there is no need to check more than 3 individual pods in the same deployment. pick up to a representative 3 from each deployment if relevant -* if you find errors and warning in a pods logs and you believe they indicate a real issue. consider the pod as not healthy. - -{% include '_toolsets_instructions.jinja2' %} - -Style guide: -* Be painfully concise. -* Leave out "the" and filler words when possible. -* your answer should ONLY return a json object with the following schema as a result: -{ - "type": "object", - "properties": { - "root_cause_summary": { - "type": "string", - "description": "concise short explaination leading to the workload_healthy result, pinpoint reason and root cause for the workload issues if any." - }, - "workload_healthy": { - "type": "boolean", - "description": "is the workload in healthy state or in error state" - } - }, - "required": [ - "root_cause_summary", - "workload_healthy" - ] -} - - -{% if alerts %} -Here are issues and configuration changes that happend to this kubernetes workload in recent time. Check if these can help you understand the issue. -{% for a in alerts %} -{{ a }} -{% endfor %} -{% endif %} diff --git a/holmes/plugins/prompts/kubernetes_workload_chat.jinja2 b/holmes/plugins/prompts/kubernetes_workload_chat.jinja2 deleted file mode 100644 index cc63b18dbf..0000000000 --- a/holmes/plugins/prompts/kubernetes_workload_chat.jinja2 +++ /dev/null @@ -1,38 +0,0 @@ -You are a tool-calling AI assist provided with common DevOps and IT tools that you can use to troubleshoot problems or answer questions. -Whenever possible, you MUST first use tools to investigate, then answer the question. -Do not say 'based on the tool output' or explicitly refer to tools at all. -If you output an answer and then realize you need to call more tools or there are possible next steps, you may do so by calling tools at that point in time. - -### Context Awareness: -Be aware that this conversation is follow-up questions to a prior investigation conducted for the {{resource}}. -However, not all questions may be directly related to that investigation. -Use results of the investigation and conversation history to maintain continuity when appropriate, ensuring efficiency in your responses. - -#### Results of Workload Health Check Analysis: -{{workload_analysis}} - -{% if tools_called_for_workload %} -Tools used for the workload analysis: -{% for tool in tools_called_for_workload %} - {{ tool }} -{% endfor %} -{% endif %} - - -{% include '_global_instructions.jinja2' %} -{% include '_general_instructions.jinja2' %} - -Style guide: -* Reply with terse output. -* Be painfully concise. -* Leave out "the" and filler words when possible. -* Be terse but not at the expense of leaving out important data like the root cause and how to fix. - -Examples: - -User: Why did the workload-example app crash? -(Call tool kubectl_find_resource kind=pod keyword=workload`) -(Call tool kubectl_previous_logs namespace=demos pod=workload-example-1299492-d9g9d # this pod name was found from the previous tool call) - -AI: `workload-example-1299492-d9g9d` crashed due to email validation error during HTTP request for /api/create_user -Relevant logs: diff --git a/server.py b/server.py index cd0c7654db..64dc907825 100644 --- a/server.py +++ b/server.py @@ -11,8 +11,6 @@ # IMPORTING ABOVE MIGHT INITIALIZE AN HTTPS CLIENT THAT DOESN'T TRUST THE CUSTOM CERTIFICATE import json from typing import List, Optional -from holmes.utils.global_instructions import generate_runbooks_args -from holmes.core.prompt import generate_user_prompt import litellm import sentry_sdk from holmes import get_version, is_official_release @@ -44,21 +42,14 @@ from holmes.core.conversations import ( build_chat_messages, build_issue_chat_messages, - build_workload_health_chat_messages, ) from holmes.core.models import ( FollowUpAction, - InvestigationResult, InvestigateRequest, - WorkloadHealthRequest, ChatRequest, ChatResponse, IssueChatRequest, - WorkloadHealthChatRequest, - workload_health_structured_output, ) -from holmes.core.investigation_structured_output import clear_json_markdown -from holmes.plugins.prompts import load_and_render_prompt from holmes.utils.holmes_sync_toolsets import holmes_sync_toolsets_status from holmes.utils.log import EndpointFilter # removed: add_runbooks_to_user_prompt @@ -208,104 +199,6 @@ def stream_investigate_issues(req: InvestigateRequest): raise HTTPException(status_code=500, detail=str(e)) -@app.post("/api/workload_health_check") -def workload_health_check(request: WorkloadHealthRequest): - try: - runbooks = config.get_runbook_catalog() - resource = request.resource - workload_alerts: list[str] = [] - if request.alert_history: - workload_alerts = dal.get_workload_issues( - resource, request.alert_history_since_hours - ) - - issue_instructions = request.instructions or [] - stored_instructions = None - if request.stored_instrucitons: - stored_instructions = dal.get_resource_instructions( - resource.get("kind", "").lower(), resource.get("name") - ) - - global_instructions = dal.get_global_instructions_for_account() - - runbooks_ctx = generate_runbooks_args( - runbook_catalog=runbooks, - global_instructions=global_instructions, - issue_instructions=issue_instructions, - resource_instructions=stored_instructions, - ) - request.ask = generate_user_prompt( - request.ask, - runbooks_ctx, - ) - ai = config.create_toolcalling_llm(dal=dal, model=request.model) - - system_prompt = load_and_render_prompt( - request.prompt_template, - context={ - "alerts": workload_alerts, - "toolsets": ai.tool_executor.toolsets, - "response_format": workload_health_structured_output, - "cluster_name": config.cluster_name, - "runbooks_enabled": True if runbooks else False, - }, - ) - - ai_call = ai.prompt_call( - system_prompt, - request.ask, - HOLMES_POST_PROCESSING_PROMPT, - workload_health_structured_output, - ) - - ai_call.result = clear_json_markdown(ai_call.result) - - return InvestigationResult( - analysis=ai_call.result, - tool_calls=ai_call.tool_calls, - instructions=issue_instructions, - metadata=ai_call.metadata, - ) - except AuthenticationError as e: - raise HTTPException(status_code=401, detail=e.message) - except litellm.exceptions.RateLimitError as e: - raise HTTPException(status_code=429, detail=e.message) - except Exception as e: - logging.exception(f"Error in /api/workload_health_check: {e}", exc_info=True) - raise HTTPException(status_code=500, detail=str(e)) - - -@app.post("/api/workload_health_chat") -def workload_health_conversation( - request: WorkloadHealthChatRequest, -): - try: - ai = config.create_toolcalling_llm(dal=dal, model=request.model) - global_instructions = dal.get_global_instructions_for_account() - - messages = build_workload_health_chat_messages( - workload_health_chat_request=request, - ai=ai, - config=config, - global_instructions=global_instructions, - ) - llm_call = ai.messages_call(messages=messages) - - return ChatResponse( - analysis=llm_call.result, - tool_calls=llm_call.tool_calls, - conversation_history=llm_call.messages, - metadata=llm_call.metadata, - ) - except AuthenticationError as e: - raise HTTPException(status_code=401, detail=e.message) - except litellm.exceptions.RateLimitError as e: - raise HTTPException(status_code=429, detail=e.message) - except Exception as e: - logging.error(f"Error in /api/workload_health_chat: {e}", exc_info=True) - raise HTTPException(status_code=500, detail=str(e)) - - @app.post("/api/issue_chat") def issue_conversation(issue_chat_request: IssueChatRequest): try: diff --git a/tests/core/test_prompt.py b/tests/core/test_prompt.py index 668e3209cb..fb8d2df877 100644 --- a/tests/core/test_prompt.py +++ b/tests/core/test_prompt.py @@ -18,15 +18,12 @@ from holmes.core.conversations import ( build_chat_messages, build_issue_chat_messages, - build_workload_health_chat_messages, ) from holmes.core.investigation import get_investigation_context from holmes.core.models import ( IssueChatRequest, - WorkloadHealthChatRequest, InvestigateRequest, IssueInvestigationResult, - WorkloadHealthInvestigationResult, ) from holmes.utils.global_instructions import generate_runbooks_args @@ -83,7 +80,6 @@ def mock_dal(): dal.get_global_instructions_for_account = Mock(return_value=None) dal.get_resource_instructions = Mock(return_value=None) dal.get_issue_data = Mock(return_value=None) - dal.get_workload_issues = Mock(return_value=[]) return dal @@ -164,22 +160,6 @@ def create_issue_chat_request(user_ask: str, issue_type: str = "prometheus"): ) -def create_workload_health_chat_request(user_ask: str, resource: Optional[dict] = None): - """Create a WorkloadHealthChatRequest for testing.""" - if resource is None: - resource = {"kind": "Deployment", "name": "my-app"} - - return WorkloadHealthChatRequest( - ask=user_ask, - conversation_history=None, - workload_health_result=WorkloadHealthInvestigationResult( - analysis="Workload is healthy", - tools=[], - ), - resource=resource, - ) - - def assert_user_prompt_contains_timestamp(user_prompt: str): """Assert that user prompt contains the UTC timestamp in seconds.""" timestamp_pattern = r"The current UTC timestamp in seconds is (\d+)\." @@ -409,55 +389,6 @@ def test_issue_chat_api_user_prompt( expected_global_instructions=extract_instructions(global_instructions), ) - def test_workload_health_chat_user_prompt(self, mock_ai, mock_config): - """Test user prompt in /api/workload_health_chat flow.""" - user_ask = "Why is my pod unhealthy?" - workload_health_chat_request = create_workload_health_chat_request(user_ask) - - messages = build_workload_health_chat_messages( - workload_health_chat_request=workload_health_chat_request, - ai=mock_ai, - config=mock_config, - global_instructions=None, - runbooks=None, - ) - - user_content = get_user_message_from_messages(messages) - validate_user_prompt(user_content, user_ask) - - @pytest.mark.parametrize( - "user_ask,global_instructions,issue_instructions", - [ - ("Check health", None, None), - ( - "Check health with instructions", - DummyInstructions(["Verify replicas"]), - ["Check pod status"], - ), - ], - ) - def test_workload_health_check_user_prompt( - self, - user_ask, - global_instructions, - issue_instructions, - ): - """Test user prompt in /api/workload_health_check flow.""" - runbooks_ctx = generate_runbooks_args( - runbook_catalog=None, - global_instructions=global_instructions, - issue_instructions=issue_instructions, - ) - - final_prompt = generate_user_prompt(user_ask, runbooks_ctx) - - validate_user_prompt( - final_prompt, - user_ask, - expected_global_instructions=extract_instructions(global_instructions), - expected_issue_instructions=issue_instructions, - ) - class TestInvestigationFlow: """Test user prompt validation for investigation flow.""" diff --git a/tests/llm/conftest.py b/tests/llm/conftest.py index 11e9b93213..ae9f10e168 100644 --- a/tests/llm/conftest.py +++ b/tests/llm/conftest.py @@ -46,7 +46,7 @@ # Configuration constants DEBUG_SEPARATOR = "=" * 80 -LLM_TEST_TYPES = ["test_ask_holmes", "test_investigate", "test_workload_health"] +LLM_TEST_TYPES = ["test_ask_holmes", "test_investigate"] def is_llm_test(nodeid: str) -> bool: @@ -55,7 +55,6 @@ def is_llm_test(nodeid: str) -> bool: [ "test_ask_holmes" in nodeid, "test_investigate" in nodeid, - "test_workload_health" in nodeid, ] ) @@ -722,8 +721,6 @@ def _collect_test_results_from_stats(terminalreporter): test_type = "ask" elif "test_investigate" in nodeid: test_type = "investigate" - elif "test_workload_health" in nodeid: - test_type = "workload_health" else: test_type = "unknown" @@ -807,8 +804,6 @@ def _collect_test_results_from_stats(terminalreporter): test_type = "ask" elif "test_investigate" in nodeid: test_type = "investigate" - elif "test_workload_health" in nodeid: - test_type = "workload_health" else: test_type = "unknown" @@ -879,7 +874,7 @@ def _collect_test_results_from_stats(terminalreporter): results_with_ids = [] for result in test_results.values(): # If we have a clean test case ID from the test, use it - # This is set in test_ask_holmes.py, test_investigate.py, and test_workload_health.py + # This is set in test_ask_holmes.py and test_investigate.py # via: request.node.user_properties.append(("clean_test_case_id", test_case.id)) # It provides the clean test case ID without model suffixes that pytest adds when # parameterizing with multiple models (e.g., "01_how_many_pods" instead of diff --git a/tests/llm/fixtures/test_workload_health/01_crashpod/issue_data.json b/tests/llm/fixtures/test_workload_health/01_crashpod/issue_data.json deleted file mode 100644 index 0967ef424b..0000000000 --- a/tests/llm/fixtures/test_workload_health/01_crashpod/issue_data.json +++ /dev/null @@ -1 +0,0 @@ -{} diff --git a/tests/llm/fixtures/test_workload_health/01_crashpod/kubectl_describe.txt b/tests/llm/fixtures/test_workload_health/01_crashpod/kubectl_describe.txt deleted file mode 100644 index 01572001d1..0000000000 --- a/tests/llm/fixtures/test_workload_health/01_crashpod/kubectl_describe.txt +++ /dev/null @@ -1,43 +0,0 @@ -{"toolset_name": "kubernetes/core", "tool_name": "kubectl_describe", "match_params": {"kind": "Deployment", "name": "crashpod", "namespace": "default"}} -{"schema_version": "robusta:v1.0.0", "status": "success", "error": null, "return_code": 0, "url": null, "invocation": "kubectl_describe command", "params": {"kind": "Deployment", "name": "crashpod", "namespace": "default"}} -stdout: - "Name: crashpod -Namespace: default -CreationTimestamp: Mon, 14 Jul 2025 14:04:46 +0300 -Labels: -Annotations: deployment.kubernetes.io/revision: 1 -Selector: app=crashpod -Replicas: 1 desired | 1 updated | 1 total | 0 available | 1 unavailable -StrategyType: RollingUpdate -MinReadySeconds: 0 -RollingUpdateStrategy: 25% max unavailable, 25% max surge -Pod Template: - Labels: app=crashpod - Containers: - crashpod: - Image: busybox - Port: - Host Port: - Command: - sh - Args: - -c - wget -O - https://gist.githubusercontent.com/odyssomay/1078370/raw/35c5981f8c139bc9dc02186f187ebee61f5b9eb9/gistfile1.txt 2>/dev/null; exit 125; - Environment: - Mounts: - Volumes: - Node-Selectors: - Tolerations: -Conditions: - Type Status Reason - ---- ------ ------ - Progressing True NewReplicaSetAvailable - Available False MinimumReplicasUnavailable -OldReplicaSets: -NewReplicaSet: crashpod-9688789bc (1/1 replicas created) -Events: - Type Reason Age From Message - ---- ------ ---- ---- ------- - Normal ScalingReplicaSet 11m deployment-controller Scaled up replica set crashpod-9688789bc from 0 to 1" - -stderr: diff --git a/tests/llm/fixtures/test_workload_health/01_crashpod/kubectl_get_by_kind_in_namespace.txt b/tests/llm/fixtures/test_workload_health/01_crashpod/kubectl_get_by_kind_in_namespace.txt deleted file mode 100644 index 5c333ca604..0000000000 --- a/tests/llm/fixtures/test_workload_health/01_crashpod/kubectl_get_by_kind_in_namespace.txt +++ /dev/null @@ -1,22 +0,0 @@ -{"toolset_name": "kubernetes/core", "tool_name": "kubectl_get_by_kind_in_namespace", "match_params": {"kind": "Pod", "namespace": "default"}} -{"schema_version": "robusta:v1.0.0", "status": "success", "error": null, "return_code": 0, "url": null, "invocation": "kubectl_get_by_kind_in_namespace command", "params": {"kind": "Pod", "namespace": "default"}} -stdout: - "NAME READY STATUS RESTARTS AGE IP NODE NOMINATED NODE READINESS GATES LABELS -aws-cloudwatch-mcp-6f756749bd-6rvgn 1/1 Running 0 3h44m 10.0.218.59 ip-10-0-203-129.us-east-2.compute.internal app=aws-cloudwatch-mcp,pod-template-hash=6f756749bd -crashpod-9688789bc-lgjzj 0/1 CrashLoopBackOff 7 (32s ago) 11m 10.0.219.148 ip-10-0-203-129.us-east-2.compute.internal app=crashpod,pod-template-hash=9688789bc -dynatrace-mcp 1/1 Running 0 18d 10.0.9.227 ip-10-0-11-226.us-east-2.compute.internal app=dynatrace-mcp -robusta-forwarder-5d54dbffd6-m8j4s 1/1 Running 0 3d20h 10.0.207.76 ip-10-0-217-140.us-east-2.compute.internal app=robusta-forwarder,pod-template-hash=5d54dbffd6 -robusta-holmes-64f57fd444-6cqng 0/1 Completed 0 3d20h 10.0.219.148 ip-10-0-203-129.us-east-2.compute.internal app=holmes,pod-template-hash=64f57fd444 -robusta-holmes-787bb4fc99-4rvh8 1/1 Running 0 3d18h 10.0.208.232 ip-10-0-217-140.us-east-2.compute.internal app=holmes,pod-template-hash=787bb4fc99 -robusta-kube-prometheus-st-admission-create-gvwmn 0/1 Completed 0 3d20h 10.0.58.139 ip-10-0-48-51.us-east-2.compute.internal app.kubernetes.io/instance=robusta,app.kubernetes.io/managed-by=Helm,app.kubernetes.io/part-of=kube-prometheus-stack,app.kubernetes.io/version=55.7.0,app=kube-prometheus-stack-admission-create,batch.kubernetes.io/controller-uid=2dc462e7-be1c-4017-a51a-48264a7069fa,batch.kubernetes.io/job-name=robusta-kube-prometheus-st-admission-create,chart=kube-prometheus-stack-55.7.0,controller-uid=2dc462e7-be1c-4017-a51a-48264a7069fa,heritage=Helm,job-name=robusta-kube-prometheus-st-admission-create,release=robusta -robusta-runner-5df9c6797f-d56p9 1/1 Running 0 3d20h 10.0.255.71 ip-10-0-228-183.us-east-2.compute.internal app=robusta-runner,pod-template-hash=5df9c6797f,robustaComponent=runner -robusta-runner-76f4b87775-hn9cn 0/1 Terminating 0 3d20h ip-10-0-48-51.us-east-2.compute.internal app=robusta-runner,pod-template-hash=76f4b87775,robustaComponent=runner -splunk-otel-collector-agent-2zhjs 1/1 Running 0 3d20h 10.0.48.51 ip-10-0-48-51.us-east-2.compute.internal app=splunk-otel-collector,component=otel-collector-agent,controller-revision-hash=6bbb959954,pod-template-generation=5,release=splunk-otel-collector -splunk-otel-collector-agent-6qz56 1/1 Running 0 5d2h 10.0.203.129 ip-10-0-203-129.us-east-2.compute.internal app=splunk-otel-collector,component=otel-collector-agent,controller-revision-hash=6bbb959954,pod-template-generation=5,release=splunk-otel-collector -splunk-otel-collector-agent-9n7t9 1/1 Running 0 5d2h 10.0.217.140 ip-10-0-217-140.us-east-2.compute.internal app=splunk-otel-collector,component=otel-collector-agent,controller-revision-hash=6bbb959954,pod-template-generation=5,release=splunk-otel-collector -splunk-otel-collector-agent-fd7dv 1/1 Running 0 5d2h 10.0.11.226 ip-10-0-11-226.us-east-2.compute.internal app=splunk-otel-collector,component=otel-collector-agent,controller-revision-hash=6bbb959954,pod-template-generation=5,release=splunk-otel-collector -splunk-otel-collector-agent-k9vgs 1/1 Running 0 5d2h 10.0.228.183 ip-10-0-228-183.us-east-2.compute.internal app=splunk-otel-collector,component=otel-collector-agent,controller-revision-hash=6bbb959954,pod-template-generation=5,release=splunk-otel-collector -splunk-otel-collector-agent-xt6kz 1/1 Running 0 5d2h 10.0.55.20 ip-10-0-55-20.us-east-2.compute.internal app=splunk-otel-collector,component=otel-collector-agent,controller-revision-hash=6bbb959954,pod-template-generation=5,release=splunk-otel-collector -splunk-otel-collector-k8s-cluster-receiver-6b9c994495-jdbmk 1/1 Running 0 5d2h 10.0.216.172 ip-10-0-228-183.us-east-2.compute.internal app=splunk-otel-collector,component=otel-k8s-cluster-receiver,pod-template-hash=6b9c994495,release=splunk-otel-collector", - -stderr: diff --git a/tests/llm/fixtures/test_workload_health/01_crashpod/kubectl_logs_all_containers.txt b/tests/llm/fixtures/test_workload_health/01_crashpod/kubectl_logs_all_containers.txt deleted file mode 100644 index e6465c31f0..0000000000 --- a/tests/llm/fixtures/test_workload_health/01_crashpod/kubectl_logs_all_containers.txt +++ /dev/null @@ -1,120 +0,0 @@ -{"toolset_name": "kubernetes/logs", "tool_name": "fetch_pod_logs", "match_params": {"pod_name": "crashpod-9688789bc-lgjzj", "namespace": "default"}} -{"schema_version": "robusta:v1.0.0", "status": "success", "error": null, "return_code": 0, "url": null, "invocation": "fetch_pod_logs command", "params": {"pod_name": "crashpod-9688789bc-lgjzj", "namespace": "default"}} -stdout: -"data": "Exception in thread \"AWT-EventQueue-0\" java.lang.StackOverflowError -\tat java.util.IdentityHashMap.get(IdentityHashMap.java:331) -\tat javax.swing.RepaintManager.extendDirtyRegion(RepaintManager.java:576) -\tat javax.swing.RepaintManager.addDirtyRegion0(RepaintManager.java:404) -\tat javax.swing.RepaintManager.addDirtyRegion(RepaintManager.java:468) -\tat javax.swing.JComponent.repaint(JComponent.java:4736) -\tat java.awt.Component.repaint(Component.java:3117) -\tat javax.swing.text.DefaultCaret.repaint(DefaultCaret.java:245) -\tat javax.swing.text.DefaultCaret.changeCaretPosition(DefaultCaret.java:1261) -\tat javax.swing.text.DefaultCaret.handleSetDot(DefaultCaret.java:1170) -\tat javax.swing.text.DefaultCaret.setDot(DefaultCaret.java:1151) -\tat javax.swing.text.DefaultCaret.setDot(DefaultCaret.java:1048) -\tat javax.swing.text.JTextComponent.setCaretPosition(JTextComponent.java:1666) -\tat sun.reflect.GeneratedMethodAccessor25.invoke(Unknown Source) -\tat sun.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) -\tat java.lang.reflect.Method.invoke(Method.java:616) -\tat clojure.lang.Reflector.invokeMatchingMethod(Reflector.java:90) -\tat clojure.lang.Reflector.invokeInstanceMethod(Reflector.java:28) -\tat llama.repl$repl_key_listener$update_caret_BANG___4893.invoke(repl.clj:55) -\tat llama.repl$repl_key_listener$reify__4895.caretUpdate(repl.clj:58) -\tat javax.swing.text.JTextComponent.fireCaretUpdate(JTextComponent.java:405) -\tat javax.swing.text.JTextComponent$MutableCaretEvent.fire(JTextComponent.java:4401) -\tat javax.swing.text.JTextComponent$MutableCaretEvent.stateChanged(JTextComponent.java:4423) -\tat javax.swing.text.DefaultCaret.fireStateChanged(DefaultCaret.java:799) -\tat javax.swing.text.DefaultCaret.changeCaretPosition(DefaultCaret.java:1274) -\tat javax.swing.text.DefaultCaret.handleSetDot(DefaultCaret.java:1170) -\tat javax.swing.text.DefaultCaret.setDot(DefaultCaret.java:1151) -\tat javax.swing.text.DefaultCaret.setDot(DefaultCaret.java:1048) -\tat javax.swing.text.JTextComponent.setCaretPosition(JTextComponent.java:1666) -\tat sun.reflect.GeneratedMethodAccessor25.invoke(Unknown Source) -\tat sun.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) -\tat java.lang.reflect.Method.invoke(Method.java:616) -\tat clojure.lang.Reflector.invokeMatchingMethod(Reflector.java:90) -\tat clojure.lang.Reflector.invokeInstanceMethod(Reflector.java:28) -\tat llama.repl$repl_key_listener$update_caret_BANG___4893.invoke(repl.clj:55) -\tat llama.repl$repl_key_listener$reify__4895.caretUpdate(repl.clj:58) -\tat javax.swing.text.JTextComponent.fireCaretUpdate(JTextComponent.java:405) -\tat javax.swing.text.JTextComponent$MutableCaretEvent.fire(JTextComponent.java:4401) -\tat javax.swing.text.JTextComponent$MutableCaretEvent.stateChanged(JTextComponent.java:4423) -\tat javax.swing.text.DefaultCaret.fireStateChanged(DefaultCaret.java:799) -\tat javax.swing.text.DefaultCaret.changeCaretPosition(DefaultCaret.java:1274) -\tat javax.swing.text.DefaultCaret.handleSetDot(DefaultCaret.java:1170) -\tat javax.swing.text.DefaultCaret.setDot(DefaultCaret.java:1151) -\tat javax.swing.text.DefaultCaret.setDot(DefaultCaret.java:1048) -\tat javax.swing.text.JTextComponent.setCaretPosition(JTextComponent.java:1666) -\tat sun.reflect.GeneratedMethodAccessor25.invoke(Unknown Source) -\tat sun.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) -\tat java.lang.reflect.Method.invoke(Method.java:616) -\tat clojure.lang.Reflector.invokeMatchingMethod(Reflector.java:90) -\tat clojure.lang.Reflector.invokeInstanceMethod(Reflector.java:28) -\tat llama.repl$repl_key_listener$update_caret_BANG___4893.invoke(repl.clj:55) -\tat llama.repl$repl_key_listener$reify__4895.caretUpdate(repl.clj:58) -\tat javax.swing.text.JTextComponent.fireCaretUpdate(JTextComponent.java:405) -\tat javax.swing.text.JTextComponent$MutableCaretEvent.fire(JTextComponent.java:4401) -\tat javax.swing.text.JTextComponent$MutableCaretEvent.stateChanged(JTextComponent.java:4423) -\tat javax.swing.text.DefaultCaret.fireStateChanged(DefaultCaret.java:799) -\tat javax.swing.text.DefaultCaret.changeCaretPosition(DefaultCaret.java:1274) -\tat javax.swing.text.DefaultCaret.handleSetDot(DefaultCaret.java:1170) -\tat javax.swing.text.DefaultCaret.setDot(DefaultCaret.java:1151) -\tat javax.swing.text.DefaultCaret.setDot(DefaultCaret.java:1048) -\tat javax.swing.text.JTextComponent.setCaretPosition(JTextComponent.java:1666) -\tat sun.reflect.GeneratedMethodAccessor25.invoke(Unknown Source) -\tat sun.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) -\tat java.lang.reflect.Method.invoke(Method.java:616) -\tat clojure.lang.Reflector.invokeMatchingMethod(Reflector.java:90) -\tat clojure.lang.Reflector.invokeInstanceMethod(Reflector.java:28) -\tat llama.repl$repl_key_listener$update_caret_BANG___4893.invoke(repl.clj:55) -\tat llama.repl$repl_key_listener$reify__4895.caretUpdate(repl.clj:58) -\tat javax.swing.text.JTextComponent.fireCaretUpdate(JTextComponent.java:405) -\tat javax.swing.text.JTextComponent$MutableCaretEvent.fire(JTextComponent.java:4401) -\tat javax.swing.text.JTextComponent$MutableCaretEvent.stateChanged(JTextComponent.java:4423) -\tat javax.swing.text.DefaultCaret.fireStateChanged(DefaultCaret.java:799) -\tat javax.swing.text.DefaultCaret.changeCaretPosition(DefaultCaret.java:1274) -\tat javax.swing.text.DefaultCaret.handleSetDot(DefaultCaret.java:1170) -\tat javax.swing.text.DefaultCaret.setDot(DefaultCaret.java:1151) -\tat javax.swing.text.DefaultCaret.setDot(DefaultCaret.java:1048) -\tat javax.swing.text.JTextComponent.setCaretPosition(JTextComponent.java:1666) -\tat sun.reflect.GeneratedMethodAccessor25.invoke(Unknown Source) -\tat sun.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) -\tat java.lang.reflect.Method.invoke(Method.java:616) -\tat clojure.lang.Reflector.invokeMatchingMethod(Reflector.java:90) -\tat clojure.lang.Reflector.invokeInstanceMethod(Reflector.java:28) -\tat llama.repl$repl_key_listener$update_caret_BANG___4893.invoke(repl.clj:55) -\tat llama.repl$repl_key_listener$reify__4895.caretUpdate(repl.clj:58) -\tat javax.swing.text.JTextComponent.fireCaretUpdate(JTextComponent.java:405) -\tat javax.swing.text.JTextComponent$MutableCaretEvent.fire(JTextComponent.java:4401) -\tat javax.swing.text.JTextComponent$MutableCaretEvent.stateChanged(JTextComponent.java:4423) -\tat javax.swing.text.DefaultCaret.fireStateChanged(DefaultCaret.java:799) -\tat javax.swing.text.DefaultCaret.changeCaretPosition(DefaultCaret.java:1274) -\tat javax.swing.text.DefaultCaret.handleSetDot(DefaultCaret.java:1170) -\tat javax.swing.text.DefaultCaret.setDot(DefaultCaret.java:1151) -\tat javax.swing.text.DefaultCaret.setDot(DefaultCaret.java:1048) -\tat javax.swing.text.JTextComponent.setCaretPosition(JTextComponent.java:1666) -\tat sun.reflect.GeneratedMethodAccessor25.invoke(Unknown Source) -\tat sun.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) -\tat java.lang.reflect.Method.invoke(Method.java:616) -\tat clojure.lang.Reflector.invokeMatchingMethod(Reflector.java:90) -\tat clojure.lang.Reflector.invokeInstanceMethod(Reflector.java:28) -\tat llama.repl$repl_key_listener$update_caret_BANG___4893.invoke(repl.clj:55) -\tat llama.repl$repl_key_listener$reify__4895.caretUpdate(repl.clj:58) -\tat javax.swing.text.JTextComponent.fireCaretUpdate(JTextComponent.java:405) -\tat javax.swing.text.JTextComponent$MutableCaretEvent.fire(JTextComponent.java:4401) -\tat javax.swing.text.JTextComponent$MutableCaretEvent.stateChanged(JTextComponent.java:4423) -\tat javax.swing.text.DefaultCaret.fireStateChanged(DefaultCaret.java:799) -\tat javax.swing.text.DefaultCaret.changeCaretPosition(DefaultCaret.java:1274) -\tat javax.swing.text.DefaultCaret.handleSetDot(DefaultCaret.java:1170) -\tat javax.swing.text.DefaultCaret.setDot(DefaultCaret.java:1151) -\tat javax.swing.text.DefaultCaret.setDot(DefaultCaret.java:1048) -\tat javax.swing.text.JTextComponent.setCaretPosition(JTextComponent.java:1666) -\tat sun.reflect.GeneratedMethodAccessor25.invoke(Unknown Source) -\tat sun.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) -\tat java.lang.reflect.Method.invoke(Method.java:616) -\tat clojure.lang.Reflector.invokeMatchingMethod(Reflector.java:90) -\tat clojure.lang.Reflector.invokeInstanceMethod(Reflector.java:28) -\tat llama.repl$repl_key_listener$update_caret_BANG___4893.invoke(repl.clj:55) -\tat llama.repl$repl_key_listener$reify__4895.caretUpdate(repl.clj:58) -stderr: diff --git a/tests/llm/fixtures/test_workload_health/01_crashpod/resource_instructions.json b/tests/llm/fixtures/test_workload_health/01_crashpod/resource_instructions.json deleted file mode 100644 index 0967ef424b..0000000000 --- a/tests/llm/fixtures/test_workload_health/01_crashpod/resource_instructions.json +++ /dev/null @@ -1 +0,0 @@ -{} diff --git a/tests/llm/fixtures/test_workload_health/01_crashpod/test_case.yaml b/tests/llm/fixtures/test_workload_health/01_crashpod/test_case.yaml deleted file mode 100644 index f5022407ce..0000000000 --- a/tests/llm/fixtures/test_workload_health/01_crashpod/test_case.yaml +++ /dev/null @@ -1,11 +0,0 @@ -expected_output: - - The crashpod or might mention the pod 'crashpod-9688789bc-lgjzj'. Mentioning one of the reasons for a CrashLoopBackOff state. 1. due to command 'wget' exiting with code 125 2. having recursive calls 3. StackOverflowError in the application. - - workload_healthy flag should be false -before_test: | - kubectl apply -f https://gist.githubusercontent.com/robusta-lab/283609047306dc1f05cf59806ade30b6/raw - sleep 60 -after_test: | - kubectl delete -f https://gist.githubusercontent.com/robusta-lab/283609047306dc1f05cf59806ade30b6/raw -evaluation: - correctness: 1 -# Success rate 100% for 100 evals diff --git a/tests/llm/fixtures/test_workload_health/01_crashpod/workload_health_request.json b/tests/llm/fixtures/test_workload_health/01_crashpod/workload_health_request.json deleted file mode 100644 index b0328ac0be..0000000000 --- a/tests/llm/fixtures/test_workload_health/01_crashpod/workload_health_request.json +++ /dev/null @@ -1,11 +0,0 @@ -{ - "resource": { - "namespace": "default", - "kind": "Deployment", - "name": "crashpod" - }, - "namespace": "default", - "alert_history": true, - "alert_history_since_hours": 24, - "ask": "Help me diagnose an issue with the workload default/Deployment/crashpod running in my Kubernetes cluster. Can you assist with identifying potential issues and pinpoint the root cause." -} diff --git a/tests/llm/test_workload_health.py b/tests/llm/test_workload_health.py deleted file mode 100644 index 2d6ef86999..0000000000 --- a/tests/llm/test_workload_health.py +++ /dev/null @@ -1,203 +0,0 @@ -# type: ignore -import time -from pathlib import Path -from typing import Optional -import json -import pytest -from server import workload_health_check - -from holmes.core.tracing import SpanType, TracingFactory -from holmes.core.tools_utils.tool_executor import ToolExecutor -from holmes.config import Config - -from holmes.core.supabase_dal import SupabaseDal -from tests.llm.utils.classifiers import ( - evaluate_correctness, -) -from tests.llm.utils.mock_dal import MockSupabaseDal -from tests.llm.utils.mock_toolset import MockToolsetManager -from tests.llm.utils.test_case_utils import ( - Evaluation, - HealthCheckTestCase, - check_and_skip_test, - get_models, -) -from tests.llm.utils.retry_handler import retry_on_throttle -from tests.llm.utils.property_manager import ( - set_initial_properties, - set_trace_properties, - update_test_results, - handle_test_error, -) -from os import path -from unittest.mock import patch - -from tests.llm.utils.iteration_utils import get_test_cases - -TEST_CASES_FOLDER = Path( - path.abspath(path.join(path.dirname(__file__), "fixtures", "test_workload_health")) -) - - -class MockConfig(Config): - def __init__( - self, - test_case: HealthCheckTestCase, - tracer, - mock_generation_config, - request=None, - ): - super().__init__() - self._test_case = test_case - self._tracer = tracer - self._mock_generation_config = mock_generation_config - self._request = request - - def create_tool_executor(self, dal: Optional[SupabaseDal]) -> ToolExecutor: - mock = MockToolsetManager( - test_case_folder=self._test_case.folder, - mock_generation_config=self._mock_generation_config, - request=self._request, - mock_policy=getattr(self._test_case, "mock_policy", "inherit"), - mock_overrides=getattr(self._test_case, "mock_overrides", None), - ) - - # With the new file-based mock system, mocks are loaded from disk automatically - # No need to call mock_tool() anymore - return ToolExecutor(mock.toolsets) - - -def get_workload_health_test_cases(): - return get_test_cases(TEST_CASES_FOLDER) - - -@pytest.mark.llm -@pytest.mark.parametrize("model", get_models()) -@pytest.mark.parametrize("test_case", get_workload_health_test_cases()) -def test_health_check( - model: str, - test_case: HealthCheckTestCase, - caplog, - request, - mock_generation_config, - shared_test_infrastructure, # type: ignore -): - # Set initial properties early so they're available even if test fails - set_initial_properties(request, test_case, model) - - tracer = TracingFactory.create_tracer("braintrust") - metadata = {"model": model} - tracer.start_experiment(additional_metadata=metadata) - - config = MockConfig(test_case, tracer, mock_generation_config, request) - config.model = model - - mock_dal = MockSupabaseDal( - test_case_folder=Path(test_case.folder), - generate_mocks=mock_generation_config.generate_mocks, - issue_data=test_case.issue_data, - issues_metadata=None, - resource_instructions=test_case.resource_instructions, - ) - - input = test_case.workload_health_request - expected = test_case.expected_output - - result = None - with tracer.start_trace( - name=f"{test_case.id}[{model}]", span_type=SpanType.EVAL - ) as eval_span: - set_trace_properties(request, eval_span) - check_and_skip_test(test_case, request, shared_test_infrastructure) - - try: - with patch.multiple("server", dal=mock_dal, config=config): - # Note: Currently workload_health_check does not trace llm calls and the run includes the startup time of the tools - with eval_span.start_span("Holmes Run", type=SpanType.TASK.value): - start_time = time.time() - retry_enabled = request.config.getoption("retry_on_throttle", True) - result = retry_on_throttle( - workload_health_check, - input, # Pass as positional arg - request=request, # Pass pytest request fixture for user_properties - retry_enabled=retry_enabled, - test_id=test_case.id, - model=model, - ) - holmes_duration = time.time() - start_time - eval_span.log(metadata={"Holmes Duration": holmes_duration}) - - # Check for any mock errors that occurred during tool execution - # This will raise an exception if any mock data errors happened - from tests.llm.utils.mock_toolset import check_for_mock_errors - - check_for_mock_errors(request) - - assert result, "No result returned by workload_health_check()" - - # check that analysis is json parsable otherwise failed. - print(f"\n🧪 TEST: {test_case.id}") - print(f" • Model: {model}") - print(f"** ANALYSIS **\n- {result.analysis}") - json.loads(result.analysis) - output = result.analysis - - debug_expected = "\n- ".join(expected) - - print(f"** EXPECTED **\n- {debug_expected}") - correctness_eval = evaluate_correctness( - output=output, - expected_elements=expected, - parent_span=eval_span, - caplog=caplog, - evaluation_type="strict", - ) - print( - f"\n** CORRECTNESS **\nscore = {correctness_eval.score}\nrationale = {correctness_eval.metadata.get('rationale', '')}" - ) - scores = {} - scores["correctness"] = correctness_eval.score - - # Log evaluation results directly to the span - if eval_span: - # Prepare tags with model - tags = (test_case.tags or []).copy() - tags.append(f"model:{model}") - - eval_span.log( - input=input, - output=output or "", - expected=str(expected), - dataset_record_id=test_case.id, - scores=scores, - metadata={"model": model}, - tags=tags, - ) - - tools_called = ( - [t.tool_name for t in result.tool_calls] if result.tool_calls else [] - ) - print(f"\n** TOOLS CALLED **\n{tools_called}") - print(f"\n** OUTPUT **\n{output}") - print(f"\n** SCORES **\n{scores}") - - # Update test results - update_test_results(request, output, tools_called, scores, result) - - except Exception as e: - handle_test_error( - request=request, - error=e, - eval_span=eval_span, - test_case=test_case, - model=model, - result=result, - mock_generation_config=mock_generation_config, - ) - raise - - if test_case.evaluation.correctness: - expected_correctness = test_case.evaluation.correctness - if isinstance(expected_correctness, Evaluation): - expected_correctness = expected_correctness.expected_score - assert scores.get("correctness", 0) >= expected_correctness diff --git a/tests/llm/utils/mock_dal.py b/tests/llm/utils/mock_dal.py index f067fa4478..0058f1d64b 100644 --- a/tests/llm/utils/mock_dal.py +++ b/tests/llm/utils/mock_dal.py @@ -133,9 +133,6 @@ def get_global_instructions_for_account(self) -> Optional[Instructions]: return None - def get_workload_issues(self, *args) -> list: - return [] - def get_resource_recommendation( self, limit: int = 10, diff --git a/tests/llm/utils/reporting/github_reporter.py b/tests/llm/utils/reporting/github_reporter.py index c1e0abb5b5..05ada33852 100644 --- a/tests/llm/utils/reporting/github_reporter.py +++ b/tests/llm/utils/reporting/github_reporter.py @@ -40,13 +40,6 @@ def generate_markdown_report(sorted_results: List[dict]) -> Tuple[str, List[dict investigate_skipped = 0 investigate_setup_failures = 0 - workload_health_total = 0 - workload_health_passed = 0 - workload_health_regressions = 0 - workload_health_mock_failures = 0 - workload_health_skipped = 0 - workload_health_setup_failures = 0 - for result in sorted_results: status = TestStatus(result) @@ -74,18 +67,6 @@ def generate_markdown_report(sorted_results: List[dict]) -> Tuple[str, List[dict investigate_regressions += 1 elif status.is_mock_failure: investigate_mock_failures += 1 - elif result["test_type"] == "workload_health": - workload_health_total += 1 - if status.is_skipped: - workload_health_skipped += 1 - elif status.is_setup_failure: - workload_health_setup_failures += 1 - elif status.passed: - workload_health_passed += 1 - elif status.is_regression: - workload_health_regressions += 1 - elif status.is_mock_failure: - workload_health_mock_failures += 1 # Generate summary lines if ask_holmes_total > 0: @@ -106,15 +87,6 @@ def generate_markdown_report(sorted_results: List[dict]) -> Tuple[str, List[dict if investigate_mock_failures > 0: markdown += f", {investigate_mock_failures} mock failures" markdown += "\n" - if workload_health_total > 0: - markdown += f"- workload_health: {workload_health_passed}/{workload_health_total} test cases were successful, {workload_health_regressions} regressions" - if workload_health_skipped > 0: - markdown += f", {workload_health_skipped} skipped" - if workload_health_setup_failures > 0: - markdown += f", {workload_health_setup_failures} setup failures" - if workload_health_mock_failures > 0: - markdown += f", {workload_health_mock_failures} mock failures" - markdown += "\n" # Generate detailed table markdown += "\n\n| Test suite | Test case | Status |\n" @@ -137,5 +109,5 @@ def generate_markdown_report(sorted_results: List[dict]) -> Tuple[str, List[dict return ( markdown, sorted_results, - ask_holmes_regressions + investigate_regressions + workload_health_regressions, + ask_holmes_regressions + investigate_regressions, ) diff --git a/tests/llm/utils/test_case_utils.py b/tests/llm/utils/test_case_utils.py index 0421cf38bb..35da942da7 100644 --- a/tests/llm/utils/test_case_utils.py +++ b/tests/llm/utils/test_case_utils.py @@ -14,7 +14,7 @@ CLASSIFIER_MODEL, MODEL_LIST_FILE_LOCATION, ) -from holmes.core.models import InvestigateRequest, WorkloadHealthRequest +from holmes.core.models import InvestigateRequest from holmes.core.prompt import append_file_to_user_prompt from holmes.config import Config from holmes.core.llm import DefaultLLM @@ -175,14 +175,6 @@ class InvestigateTestCase(HolmesTestCase, BaseModel): request: Any = None -class HealthCheckTestCase(HolmesTestCase, BaseModel): - workload_health_request: WorkloadHealthRequest - issue_data: Optional[Dict] - resource_instructions: Optional[ResourceInstructions] - expected_sections: Optional[Dict[str, Union[List[str], bool]]] = None - request: Any = None - - def check_and_skip_test( test_case: HolmesTestCase, request=None, shared_test_infrastructure=None ) -> None: @@ -248,9 +240,6 @@ def __init__(self, test_cases_folder: Path) -> None: super().__init__() self._test_cases_folder = test_cases_folder - def load_workload_health_test_cases(self) -> List[HealthCheckTestCase]: - return cast(List[HealthCheckTestCase], self.load_test_cases()) - def load_investigate_test_cases(self) -> List[InvestigateTestCase]: return cast(List[InvestigateTestCase], self.load_test_cases()) @@ -335,18 +324,6 @@ def load_test_cases(self) -> List[HolmesTestCase]: test_case = TypeAdapter(InvestigateTestCase).validate_python( config_dict ) - elif self._test_cases_folder.name == "test_workload_health": - config_dict["workload_health_request"] = ( - load_workload_health_request(test_case_folder) - ) - config_dict["issue_data"] = load_issue_data(test_case_folder) - config_dict["resource_instructions"] = load_resource_instructions( - test_case_folder - ) - config_dict["request"] = TypeAdapter(WorkloadHealthRequest) - test_case = TypeAdapter(HealthCheckTestCase).validate_python( - config_dict - ) elif self._test_cases_folder.name == "compaction": # Compaction tests only need conversation_history and expected_output config_dict["conversation_history"] = load_conversation_history( @@ -453,19 +430,6 @@ def load_investigate_request(test_case_folder: Path) -> InvestigateRequest: ) -def load_workload_health_request(test_case_folder: Path) -> WorkloadHealthRequest: - workload_health_request_path = test_case_folder.joinpath( - Path("workload_health_request.json") - ) - if workload_health_request_path.exists(): - return TypeAdapter(WorkloadHealthRequest).validate_json( - read_file(Path(workload_health_request_path)) - ) - raise Exception( - f"Workload health test case declared in folder {str(test_case_folder)} should have an workload_health_request.json file but none is present" - ) - - def _parse_conversation_history_md_files( conversation_history_dir, ) -> None | List[Dict[str, str]]: diff --git a/tests/test_ai_safety_prompt.py b/tests/test_ai_safety_prompt.py index 61fcb38ab8..faff326186 100644 --- a/tests/test_ai_safety_prompt.py +++ b/tests/test_ai_safety_prompt.py @@ -13,7 +13,6 @@ class TestAISafetyPromptInclusion: "builtin://generic_ask.jinja2", "builtin://generic_ask_conversation.jinja2", "builtin://generic_ask_for_issue_conversation.jinja2", - "builtin://kubernetes_workload_ask.jinja2", "builtin://generic_investigation.jinja2", ], ) diff --git a/tests/test_server_endpoints.py b/tests/test_server_endpoints.py index 0a0abd923a..5a772c6092 100644 --- a/tests/test_server_endpoints.py +++ b/tests/test_server_endpoints.py @@ -132,130 +132,3 @@ def test_api_issue_chat_all_fields( assert "tool_name" in tool_call assert "description" in tool_call assert "result" in tool_call - - -@patch("holmes.config.Config.create_toolcalling_llm") -@patch("holmes.core.supabase_dal.SupabaseDal.get_global_instructions_for_account") -def test_api_workload_health_chat( - mock_get_global_instructions, - mock_create_toolcalling_llm, - client, -): - mock_ai = MagicMock() - mock_ai.messages_call.return_value = MagicMock( - result="This is a mock analysis for workload health chat.", - tool_calls=[ - { - "tool_call_id": "1", - "tool_name": "health_checker", - "description": "Checks workload health", - "result": {"status": "success", "data": "Workload is healthy"}, - } - ], - messages=[ - {"role": "system", "content": "You are a helpful assistant."}, - {"role": "user", "content": "Check the workload health."}, - ], - metadata={}, - ) - mock_create_toolcalling_llm.return_value = mock_ai - - mock_get_global_instructions.return_value = [] - - payload = { - "ask": "Check the workload health.", - "workload_health_result": { - "analysis": "Mock workload health analysis", - "tools": [], - }, - "resource": {"name": "example-resource", "kind": "Deployment"}, - "conversation_history": [ - {"role": "system", "content": "You are a helpful assistant."}, - {"role": "user", "content": "Check the workload health."}, - ], - } - response = client.post("/api/workload_health_chat", json=payload) - assert response.status_code == 200 - data = response.json() - - assert "analysis" in data - assert "conversation_history" in data - assert "tool_calls" in data - - assert isinstance(data["analysis"], str) - assert isinstance(data["conversation_history"], list) - assert isinstance(data["tool_calls"], list) - - assert any(msg.get("role") == "user" for msg in data["conversation_history"]) - - if data["tool_calls"]: - tool_call = data["tool_calls"][0] - assert "tool_call_id" in tool_call - assert "tool_name" in tool_call - assert "description" in tool_call - assert "result" in tool_call - - -@patch("holmes.config.Config.create_toolcalling_llm") -@patch("holmes.core.supabase_dal.SupabaseDal.get_global_instructions_for_account") -@patch("holmes.core.supabase_dal.SupabaseDal.get_workload_issues") -@patch("holmes.core.supabase_dal.SupabaseDal.get_resource_instructions") -@patch("holmes.plugins.prompts.load_and_render_prompt") -def test_api_workload_health_check( - mock_load_and_render_prompt, - mock_get_resource_instructions, - mock_get_workload_issues, - mock_get_global_instructions, - mock_create_toolcalling_llm, - client, -): - mock_ai = MagicMock() - mock_ai.prompt_call.return_value = MagicMock( - result="This is a mock analysis for workload health check.", - tool_calls=[ - { - "tool_call_id": "1", - "tool_name": "health_checker", - "description": "Checks workload health", - "result": {"status": "success", "data": "Workload is healthy"}, - } - ], - metadata={}, - ) - mock_create_toolcalling_llm.return_value = mock_ai - - mock_get_global_instructions.return_value = [] - mock_get_workload_issues.return_value = ["Alert 1", "Alert 2"] - mock_get_resource_instructions.return_value = MagicMock( - instructions=["Instruction 1", "Instruction 2"] - ) - - mock_load_and_render_prompt.return_value = "Mocked system prompt" - - payload = { - "resource": {"name": "example-resource", "kind": "Deployment"}, - "alert_history": True, - "alert_history_since_hours": 24, - "instructions": ["Check CPU usage", "Check memory usage"], - "stored_instructions": True, - "ask": "Check the workload health.", - "model": "gpt-4.1", - } - response = client.post("/api/workload_health_check", json=payload) - assert response.status_code == 200 - data = response.json() - - assert "analysis" in data - assert "tool_calls" in data - assert "instructions" in data - - assert isinstance(data["analysis"], str) - assert isinstance(data["tool_calls"], list) - assert isinstance(data["instructions"], list) - - if data["tool_calls"]: - tool_call = data["tool_calls"][0] - assert "tool_call_id" in tool_call - assert "tool_name" in tool_call - assert "description" in tool_call - assert "result" in tool_call