From 4000d46cbf596df9bac50671282b7a44280a7415 Mon Sep 17 00:00:00 2001 From: Arik Alon Date: Wed, 6 Aug 2025 20:25:03 +0300 Subject: [PATCH 01/17] subtasks commit --- holmes/core/investigation.py | 8 + holmes/core/prompt.py | 7 + holmes/core/todo_manager.py | 118 ++++++++ holmes/core/tool_calling_llm.py | 11 + holmes/core/tools.py | 135 +++++++++ .../prompts/_general_instructions.jinja2 | 12 + holmes/plugins/prompts/generic_ask.jinja2 | 24 ++ .../prompts/generic_investigation.jinja2 | 28 ++ holmes/plugins/toolsets/__init__.py | 4 + .../plugins/toolsets/investigator/__init__.py | 0 .../investigator/core_investigation.py | 32 ++ .../investigator_instructions.jinja2 | 273 ++++++++++++++++++ tests/core/test_todo_manager.py | 108 +++++++ tests/core/test_todo_write_tool.py | 183 ++++++++++++ .../toolsets/test_core_investigation.py | 41 +++ 15 files changed, 984 insertions(+) create mode 100644 holmes/core/todo_manager.py create mode 100644 holmes/plugins/toolsets/investigator/__init__.py create mode 100644 holmes/plugins/toolsets/investigator/core_investigation.py create mode 100644 holmes/plugins/toolsets/investigator/investigator_instructions.jinja2 create mode 100644 tests/core/test_todo_manager.py create mode 100644 tests/core/test_todo_write_tool.py create mode 100644 tests/plugins/toolsets/test_core_investigation.py diff --git a/holmes/core/investigation.py b/holmes/core/investigation.py index c06f506d27..3f1ff56f57 100644 --- a/holmes/core/investigation.py +++ b/holmes/core/investigation.py @@ -125,6 +125,13 @@ def get_investigation_context( else: logging.info("Structured output is disabled for this request") + # Add TodoList context to prompt + from holmes.core.todo_manager import get_todo_manager, get_session_id_from_context + + todo_manager = get_todo_manager() + session_id = get_session_id_from_context() + todo_context = todo_manager.format_tasks_for_prompt(session_id) + system_prompt = load_and_render_prompt( investigate_request.prompt_template, { @@ -133,6 +140,7 @@ def get_investigation_context( "structured_output": request_structured_output_from_llm, "toolsets": ai.tool_executor.toolsets, "cluster_name": config.cluster_name, + "todo_list": todo_context, }, ) diff --git a/holmes/core/prompt.py b/holmes/core/prompt.py index e2e8ef4fd0..c317658e8d 100644 --- a/holmes/core/prompt.py +++ b/holmes/core/prompt.py @@ -59,6 +59,13 @@ def build_initial_ask_messages( console, initial_user_prompt, file_paths ) + user_prompt_with_files += ( + "\n\n\nIMPORTANT: You have access to the TodoWrite tool. It creates a TodoList, in order to track progress. It's very important. You MUST use it:\n1. FIRST: Ask your self which sub problems you need to solve in order to answer the question." + "Do this, BEFORE any other tools\n2. " + "AFTER EVERY TOOL CALL: If required, update the TodoList\n3. " + "\n\nFAILURE TO UPDATE TodoList = INCOMPLETE INVESTIGATION\n\n" + "Example flow:\n- Think and divide to sub problems → create TodoList → Perform each task on the list → Update list → Verify your solution\n" + ) messages = [ {"role": "system", "content": system_prompt_rendered}, {"role": "user", "content": user_prompt_with_files}, diff --git a/holmes/core/todo_manager.py b/holmes/core/todo_manager.py new file mode 100644 index 0000000000..b80722406c --- /dev/null +++ b/holmes/core/todo_manager.py @@ -0,0 +1,118 @@ +import logging +from typing import Dict, List +from threading import Lock +from uuid import uuid4 + +from holmes.core.tools import Task, TaskStatus + + +class TodoListManager: + """ + Session-based storage manager for investigation TodoLists. + Stores TodoLists per session and provides methods to get/update tasks. + """ + + def __init__(self): + self._sessions: Dict[str, List[Task]] = {} + self._lock = Lock() + + def get_session_tasks(self, session_id: str) -> List[Task]: + """Get all tasks for a session. Returns empty list if session doesn't exist.""" + logging.info(f"########## get_session_tasks {session_id}") + with self._lock: + return self._sessions.get(session_id, []).copy() + + def update_session_tasks(self, session_id: str, tasks: List[Task]) -> None: + """Update all tasks for a session.""" + with self._lock: + self._sessions[session_id] = tasks.copy() + + def clear_session(self, session_id: str) -> None: + """Clear all tasks for a session.""" + with self._lock: + if session_id in self._sessions: + del self._sessions[session_id] + + def get_session_count(self) -> int: + """Get number of active sessions (for debugging/testing).""" + with self._lock: + return len(self._sessions) + + def format_tasks_for_prompt(self, session_id: str) -> str: + """ + Format tasks for injection into system prompt. + Returns empty string if no tasks exist. + """ + tasks = self.get_session_tasks(session_id) + + if not tasks: + return "" + + # Sort tasks by status (pending -> in_progress -> completed) then priority + status_order = {"pending": 0, "in_progress": 1, "completed": 2} + + sorted_tasks = sorted( + tasks, + key=lambda t: (status_order.get(t.status.value, 3),), + ) + + lines = ["# CURRENT INVESTIGATION TASKS"] + lines.append("") + + # Count tasks by status + pending_count = sum(1 for t in tasks if t.status == TaskStatus.PENDING) + progress_count = sum(1 for t in tasks if t.status == TaskStatus.IN_PROGRESS) + completed_count = sum(1 for t in tasks if t.status == TaskStatus.COMPLETED) + + lines.append( + f"**Task Status**: {completed_count} completed, {progress_count} in progress, {pending_count} pending" + ) + lines.append("") + + for task in sorted_tasks: + # Use simple text indicators for prompt injection + status_indicator = { + "pending": "[ ]", + "in_progress": "[~]", + "completed": "[✓]", + }.get(task.status.value, "[?]") + + lines.append(f"{status_indicator} [{task.id}] {task.content}") + + lines.append("") + lines.append( + "**Instructions**: Use TodoWrite tool to update task status as you work. Mark tasks as 'in_progress' when starting, 'completed' when finished." + ) + + return "\n".join(lines) + + +# Global instance for session management +_todo_manager = TodoListManager() + + +def get_todo_manager() -> TodoListManager: + """Get the global TodoListManager instance.""" + return _todo_manager + + +def get_session_id_from_context(context=None) -> str: + """ + Extract or generate session ID from context. + For now, we'll use a simple approach - this can be enhanced later. + """ + # TODO: This should be enhanced to extract session ID from investigation context + # For now, we'll use a simple thread-local or context-based approach + if hasattr(context, "_current_session"): + return context._current_session + + # Generate new session ID + session_id = str(uuid4()) + context._current_session = session_id + return session_id + + +def set_current_session_id(session_id: str) -> None: + """Set the current session ID for this thread/context.""" + pass + # get_session_id_from_context._current_session = session_id diff --git a/holmes/core/tool_calling_llm.py b/holmes/core/tool_calling_llm.py index cd18a260fc..77e6183cf1 100644 --- a/holmes/core/tool_calling_llm.py +++ b/holmes/core/tool_calling_llm.py @@ -823,6 +823,16 @@ def investigate( "[bold]No runbooks found for this issue. Using default behaviour. (Add runbooks to guide the investigation.)[/bold]" ) + # Add TodoList context to prompt + from holmes.core.todo_manager import ( + get_todo_manager, + get_session_id_from_context, + ) + + todo_manager = get_todo_manager() + session_id = get_session_id_from_context() + todo_context = todo_manager.format_tasks_for_prompt(session_id) + system_prompt = load_and_render_prompt( prompt, { @@ -831,6 +841,7 @@ def investigate( "structured_output": request_structured_output_from_llm, "toolsets": self.tool_executor.toolsets, "cluster_name": self.cluster_name, + "todo_list": todo_context, }, ) diff --git a/holmes/core/tools.py b/holmes/core/tools.py index cff65e18f0..344cd47e2f 100644 --- a/holmes/core/tools.py +++ b/holmes/core/tools.py @@ -18,6 +18,7 @@ from holmes.plugins.prompts import load_and_render_prompt import time from rich.table import Table +from uuid import uuid4 class ToolResultStatus(str, Enum): @@ -569,3 +570,137 @@ def pretty_print_toolset_status(toolsets: list[Toolset], console: Console) -> No table.add_row(*(str(row.get(col.capitalize(), "")) for col in status_fields)) console.print(table) + + +class TaskStatus(str, Enum): + PENDING = "pending" + IN_PROGRESS = "in_progress" + COMPLETED = "completed" + + +class TaskPriority(str, Enum): + HIGH = "high" + MEDIUM = "medium" + LOW = "low" + + +class Task(BaseModel): + id: str = Field(default_factory=lambda: str(uuid4())) + content: str + status: TaskStatus = TaskStatus.PENDING + + +class TodoWriteTool(Tool): + name: str = "TodoWrite" + description: str = "Save investigation tasks to break down complex problems into manageable sub-tasks" + parameters: Dict[str, ToolParameter] = { + "todos": ToolParameter( + description="List of tasks to track during the investigation. Each task should have: id (string), content (string), status (pending/in_progress/completed)", + type="array[object]", + required=True, + ) + } + + def _invoke(self, params: Dict) -> StructuredToolResult: + try: + from holmes.core.todo_manager import ( + get_todo_manager, + get_session_id_from_context, + ) + + todos_data = params.get("todos", []) + + tasks = [] + + for todo_item in todos_data: + if isinstance(todo_item, dict): + task = Task( + id=todo_item.get("id", str(uuid4())), + content=todo_item.get("content", ""), + status=TaskStatus(todo_item.get("status", "pending")), + ) + tasks.append(task) + + logging.info(f"Tasks: {len(tasks)}") + + # Store tasks in session storage + todo_manager = get_todo_manager() + session_id = get_session_id_from_context() + todo_manager.update_session_tasks(session_id, tasks) + + # Print a nice table to console/log + def print_tasks_table(tasks): + if not tasks: + logging.info("No tasks in the investigation plan.") + return + + # Calculate column widths + max_id_width = max(len(str(task.id)) for task in tasks) + max_content_width = max(len(task.content) for task in tasks) + max_status_width = max(len(task.status.value) for task in tasks) + max_priority_width = max( + len(task.priority.value.upper()) for task in tasks + ) + + # Ensure minimum widths for headers + id_width = max(max_id_width, 2) + content_width = max(max_content_width, 7) + status_width = max(max_status_width, 6) + priority_width = max(max_priority_width, 8) + + # Status indicators + status_icons = { + "pending": "[ ]", + "in_progress": "[~]", + "completed": "[✓]", + } + + # Build table + separator = f"+{'-' * (id_width + 2)}+{'-' * (content_width + 2)}+{'-' * (status_width + 2)}+{'-' * (priority_width + 2)}+" + header = f"| {'ID':<{id_width}} | {'Content':<{content_width}} | {'Status':<{status_width}} | {'Priority':<{priority_width}} |" + + # Log the table + logging.info("Updated Investigation Tasks:") + logging.info(separator) + logging.info(header) + logging.info(separator) + + for task in tasks: + status_display = ( + f"{status_icons[task.status.value]} {task.status.value}" + ) + priority_display = task.priority.value.upper() + row = f"| {task.id:<{id_width}} | {task.content:<{content_width}} | {status_display:<{status_width}} | {priority_display:<{priority_width}} |" + logging.info(row) + + logging.info(separator) + + # Print table to console/log + print_tasks_table(tasks) + + # Get pretty formatted version of the updated TodoList (keep existing format) + formatted_tasks = todo_manager.format_tasks_for_prompt(session_id) + + # Return confirmation with pretty printed TodoList (unchanged) + response_data = f"✅ Investigation plan updated with {len(tasks)} tasks. Tasks are now stored in session and will appear in subsequent prompts.\n\n" + if formatted_tasks: + response_data += formatted_tasks + else: + response_data += "No tasks currently in the investigation plan." + + return StructuredToolResult( + status=ToolResultStatus.SUCCESS, + data=response_data, + params=params, + ) + + except Exception as e: + return StructuredToolResult( + status=ToolResultStatus.ERROR, + error=f"Failed to process tasks: {str(e)}", + params=params, + ) + + def get_parameterized_one_liner(self, params: Dict) -> str: + todos = params.get("todos", []) + return f"Write {todos} investigation tasks" diff --git a/holmes/plugins/prompts/_general_instructions.jinja2 b/holmes/plugins/prompts/_general_instructions.jinja2 index b9e5941b6a..a8e5b8d053 100644 --- a/holmes/plugins/prompts/_general_instructions.jinja2 +++ b/holmes/plugins/prompts/_general_instructions.jinja2 @@ -47,6 +47,18 @@ * For any question, try to make the answer specific to the user's cluster. ** For example, if asked to port forward, find out the app or pod port (kubectl describe) and provide a port forward command specific to the user's question +# MANDATORY Task Management + +* You MUST use the TodoWrite tool for ANY investigation requiring multiple steps +* Your FIRST tool call MUST be TodoWrite to create your investigation plan +* Break down ALL complex problems into smaller, manageable tasks +* You MUST update task status (pending → in_progress → completed) as you work through your investigation +* The TodoWrite tool will show you a formatted task list - reference this throughout your investigation +* Mark tasks as 'in_progress' when you start them, 'completed' when finished +* Follow ALL tasks in your plan - don't skip any tasks +* Use task management to ensure you don't miss important investigation steps +* If you discover additional steps during investigation, add them to your task list using TodoWrite + # Tool/function calls You are able to make tool calls / function calls. Recognise when a tool has already been called and reuse its result. diff --git a/holmes/plugins/prompts/generic_ask.jinja2 b/holmes/plugins/prompts/generic_ask.jinja2 index fb7d9b35c8..9103635824 100644 --- a/holmes/plugins/prompts/generic_ask.jinja2 +++ b/holmes/plugins/prompts/generic_ask.jinja2 @@ -4,6 +4,30 @@ Ask for multiple tool calls at the same time as it saves time for the user. Do not say 'based on the tool output' or explicitly refer to tools at all. If you output an answer and then realize you need to call more tools or there are possible next steps, you may do so by calling tools at that point in time. If you have a good and concrete suggestion for how the user can fix something, tell them even if not asked explicitly + +CRITICAL: For multi-step questions, you MUST start by calling the TodoWrite tool to create your investigation plan. + +{% if todo_list %} +{{ todo_list }} +{% endif %} + +# MANDATORY Investigation Process + +For ANY question that requires multiple steps or investigation, you MUST follow this structured approach: + +1. **IMMEDIATELY START with TodoWrite**: Before doing any investigation, you MUST call the TodoWrite tool to create an investigation plan. Break down the question into specific tasks. + +2. **Problem Analysis**: Identify what sub-problems need to be solved and create tasks for each. + +3. **Task Execution**: Execute your plan systematically. You MUST: + - Mark tasks as 'in_progress' when you start working on them + - Call TodoWrite to update task status as you complete each task + - Reference the visible task list from your TodoWrite output throughout your investigation + - Follow the exact tasks you created - complete all of them + +4. **Verification**: Before providing your final answer, verify that your conclusion is complete and addresses the user's question fully. + +IMPORTANT: If a question requires more than one investigation step, your FIRST tool call MUST be TodoWrite. {% include '_current_date_time.jinja2' %} Use conversation history to maintain continuity when appropriate, ensuring efficiency in your responses. diff --git a/holmes/plugins/prompts/generic_investigation.jinja2 b/holmes/plugins/prompts/generic_investigation.jinja2 index 2789932afe..d0ac1195d5 100644 --- a/holmes/plugins/prompts/generic_investigation.jinja2 +++ b/holmes/plugins/prompts/generic_investigation.jinja2 @@ -3,6 +3,34 @@ Whenever possible you MUST first use tools to investigate then answer the questi Ask for multiple tool calls at the same time as it saves time for the user. Do not say 'based on the tool output' +CRITICAL: For ALL investigations, you MUST start by calling the TodoWrite tool to create your investigation plan. + +{% if todo_list %} +{{ todo_list }} +{% endif %} + +# MANDATORY Investigation Process + +You MUST follow this structured approach for ALL investigations. NO EXCEPTIONS: + +1. **IMMEDIATELY START with TodoWrite**: Before doing ANYTHING else, you MUST call the TodoWrite tool to create an investigation plan. Analyze the issue and break it down into specific tasks. + +2. **Problem Analysis**: Ask yourself: "What are the sub-problems I need to solve to answer this question?" Create tasks for each sub-problem. + +3. **Task Execution**: Execute your investigation plan by calling the appropriate tools. You MUST: + - Mark tasks as 'in_progress' when you start working on them + - Call TodoWrite to update task status as you complete each task + - Reference the visible task list from your TodoWrite output throughout your investigation + - Follow the exact tasks you created - don't skip or ignore any + +4. **Verification**: Once you believe you have found the answer, perform a verification step: + - Cross-check your findings with the original alert/issue details + - Ensure your conclusion addresses the root cause, not just symptoms + - Verify that your analysis is consistent with all collected data + - If verification fails, use TodoWrite to add new tasks and return to investigation + +REMEMBER: Your first tool call MUST ALWAYS be TodoWrite to create your investigation plan. + Provide an terse analysis of the following {{ issue.source_type }} alert/issue and why it is firing. * {% include '_current_date_time.jinja2' %} * If the tool requires string format timestamps, query from 'start_timestamp' until 'end_timestamp' diff --git a/holmes/plugins/toolsets/__init__.py b/holmes/plugins/toolsets/__init__.py index e05aaa5753..dfe7e2f0df 100644 --- a/holmes/plugins/toolsets/__init__.py +++ b/holmes/plugins/toolsets/__init__.py @@ -39,6 +39,9 @@ from holmes.plugins.toolsets.robusta.robusta import RobustaToolset from holmes.plugins.toolsets.runbook.runbook_fetcher import RunbookToolset from holmes.plugins.toolsets.servicenow.servicenow import ServiceNowToolset +from holmes.plugins.toolsets.investigator.core_investigation import ( + CoreInvestigationToolset, +) THIS_DIR = os.path.abspath(os.path.dirname(__file__)) @@ -63,6 +66,7 @@ def load_toolsets_from_file( def load_python_toolsets(dal: Optional[SupabaseDal]) -> List[Toolset]: logging.debug("loading python toolsets") toolsets: list[Toolset] = [ + CoreInvestigationToolset(), # Load first for higher priority InternetToolset(), RobustaToolset(dal), OpenSearchToolset(), diff --git a/holmes/plugins/toolsets/investigator/__init__.py b/holmes/plugins/toolsets/investigator/__init__.py new file mode 100644 index 0000000000..e69de29bb2 diff --git a/holmes/plugins/toolsets/investigator/core_investigation.py b/holmes/plugins/toolsets/investigator/core_investigation.py new file mode 100644 index 0000000000..c3e927f99b --- /dev/null +++ b/holmes/plugins/toolsets/investigator/core_investigation.py @@ -0,0 +1,32 @@ +import logging +import os +from typing import Any, Dict + +from holmes.core.tools import TodoWriteTool, Toolset, ToolsetTag, ToolsetStatusEnum + + +class CoreInvestigationToolset(Toolset): + """Core toolset for investigation management and task planning.""" + + def __init__(self): + super().__init__( + name="core_investigation", + description="Core investigation tools for task management and planning", + enabled=True, + tools=[TodoWriteTool()], + tags=[ToolsetTag.CORE], + is_default=True, + ) + # Override the default DISABLED status to ENABLED since this is a core toolset + self.status = ToolsetStatusEnum.ENABLED + logging.info("Core investigation toolset loaded") + + def get_example_config(self) -> Dict[str, Any]: + return {} + + def _reload_instructions(self): + """Load Datadog metrics specific troubleshooting instructions.""" + template_file_path = os.path.abspath( + os.path.join(os.path.dirname(__file__), "investigator_instructions.jinja2") + ) + self._load_llm_instructions(jinja_template=f"file://{template_file_path}") diff --git a/holmes/plugins/toolsets/investigator/investigator_instructions.jinja2 b/holmes/plugins/toolsets/investigator/investigator_instructions.jinja2 new file mode 100644 index 0000000000..c8f6da4b10 --- /dev/null +++ b/holmes/plugins/toolsets/investigator/investigator_instructions.jinja2 @@ -0,0 +1,273 @@ +# Task Management +You have access to the TodoWrite tool to help you manage and plan tasks. The updated task list will always appear in the prompt. Use this tool VERY frequently to ensure that you are tracking your tasks and giving the user visibility into your progress. +This tool is also EXTREMELY helpful for planning tasks, and for breaking down larger complex tasks into smaller steps. If you do not use this tool when planning, you may forget to do important tasks - and that is unacceptable. + +It is critical that you mark todos as completed as soon as you are done with a task. Do not batch up multiple tasks before marking them as completed. + +Examples: + + +user: Run the build and fix any type errors +assistant: I'm going to use the TodoWrite tool to write the following items to the todo list: +- Run the build +- Fix any type errors + +I'm now going to run the build using Bash. + +Looks like I found 10 type errors. I'm going to use the TodoWrite tool to write 10 items to the todo list. + +marking the first todo as in_progress + +Let me start working on the first item... + +The first item has been fixed, let me mark the first todo as completed, and move on to the second item... +.. +.. + +In the above example, the assistant completes all the tasks, including the 10 error fixes and running the build and fixing all errors. + + +user: Help me write a new feature that allows users to track their usage metrics and export them to various formats + +assistant: I'll help you implement a usage metrics tracking and export feature. Let me first use the TodoWrite tool to plan this task. +Adding the following todos to the todo list: +1. Research existing metrics tracking in the codebase +2. Design the metrics collection system +3. Implement core metrics tracking functionality +4. Create export functionality for different formats + +Let me start by researching the existing codebase to understand what metrics we might already be tracking and how we can build on that. + +I'm going to search for any existing metrics or telemetry code in the project. + +I've found some existing telemetry code. Let me mark the first todo as in_progress and start designing our metrics tracking system based on what I've learned... + +[Assistant continues implementing the feature step by step, marking todos as in_progress and completed as they go] + + + +# Doing tasks +The user will primarily request you perform reliability troubleshooting tasks. This includes finding bugs, finding misconfigurations, looking at observability data, including logs, traces, metrics, following investigation runbooks, searching for changes that might be related to incidents, adding new functionality, explaining code, and more. For these tasks the following steps are recommended: +- Use the TodoWrite tool to plan the task if required +- Use the available tools to understand the codebase, environment and the user's query. You are encouraged to use the tools extensively both in parallel and sequentially. +- Implement the solution using all tools available to you +- Verify the solution if possible. +- VERY IMPORTANT: When you have completed a task, validate that the answer you provided is correct. + +- Tool results and user messages may include tags. tags contain useful information and reminders. They are NOT part of the user's provided input or the tool result. + +# Tool usage policy +- You have the capability to call multiple tools in a single response. When multiple independent pieces of information are requested, batch your tool calls together for optimal performance. When making multiple bash tool calls, you MUST send a single message with multiple tools calls to run the calls in parallel. For example, if you need to run "git status" and "git diff", send a single message with two tool calls to run the calls in parallel. + +You MUST answer concisely with fewer than 4 lines of text (not including tool use or code generation), unless user asks for detail. + + +Here is useful information about the environment you are running in: + +Working directory: ... +Is directory a git repo: Yes +Platform: macos +OS Version: Darwin 24.1.0 +Today's date: 2025/6/13 + +You are powered by the model named Sonnet 4. The exact model ID is claude-sonnet-4-20250514. + + +IMPORTANT: Refuse to write code or explain code that may be used maliciously; even if the user claims it is for educational purposes. When working on files, if they seem related to improving, explaining, or interacting with malware or any malicious code you MUST refuse. +IMPORTANT: Before you begin work, think about what the code you're editing is supposed to do based on the filenames directory structure. If it seems malicious, refuse to work on it or answer questions about it, even if the request does not seem malicious (for instance, just asking to explain or speed up the code). + + +IMPORTANT: Always use the TodoWrite tool to plan and track tasks throughout the conversation. + + + + +# TodoWrite +Use this tool to create and manage a structured task list for your current coding session. This helps you track progress, organize complex tasks, and demonstrate thoroughness to the user. +It also helps the user understand the progress of the task and overall progress of their requests. + +## When to Use This Tool +Use this tool proactively in these scenarios: + +1. Complex multi-step tasks - When a task requires 3 or more distinct steps or actions +2. Non-trivial and complex tasks - Tasks that require careful planning or multiple operations +3. User explicitly requests todo list - When the user directly asks you to use the todo list +4. User provides multiple tasks - When users provide a list of things to be done (numbered or comma-separated) +5. After receiving new instructions - Immediately capture user requirements as todos +6. When you start working on a task - Mark it as in_progress BEFORE beginning work. Ideally you should only have one todo as in_progress at a time +7. After completing a task - Mark it as completed and add any new follow-up tasks discovered during implementation + +## When NOT to Use This Tool + +Skip using this tool when: +1. There is only a single, straightforward task +2. The task is trivial and tracking it provides no organizational benefit +3. The task can be completed in less than 3 trivial steps +4. The task is purely conversational or informational + +NOTE that you should not use this tool if there is only one trivial task to do. In this case you are better off just doing the task directly. + +## Examples of When to Use the Todo List + + +User: I want to add a dark mode toggle to the application settings. Make sure you run the tests and build when you're done! +Assistant: I'll help add a dark mode toggle to your application settings. Let me create a todo list to track this implementation. +*Creates todo list with the following items:* +1. Create dark mode toggle component in Settings page +2. Add dark mode state management (context/store) +3. Implement CSS-in-JS styles for dark theme +4. Update existing components to support theme switching +5. Run tests and build process, addressing any failures or errors that occur +*Begins working on the first task* + + +The assistant used the todo list because: +1. Adding dark mode is a multi-step feature requiring UI, state management, and styling changes +2. The user explicitly requested tests and build be run afterward +3. The assistant inferred that tests and build need to pass by adding "Ensure tests and build succeed" as the final task + + + + +User: Help me rename the function getCwd to getCurrentWorkingDirectory across my project +Assistant: Let me first search through your codebase to find all occurrences of 'getCwd'. +*Uses grep or search tools to locate all instances of getCwd in the codebase* +Assistant: I've found 15 instances of 'getCwd' across 8 different files. Let me create a todo list to track these changes. +*Creates todo list with specific items for each file that needs updating* + + +The assistant used the todo list because: +1. First, the assistant searched to understand the scope of the task +2. Upon finding multiple occurrences across different files, it determined this was a complex task with multiple steps +3. The todo list helps ensure every instance is tracked and updated systematically +4. This approach prevents missing any occurrences and maintains code consistency + + + + + +User: I need to implement these features for my e-commerce site: user registration, product catalog, shopping cart, and checkout flow. +Assistant: I'll help implement these features. First, let's add all the features to the todo list. +*Creates a todo list breaking down each feature into specific tasks based on the project architecture* +Assistant: Let's start with user registration. This will involve creating database models, API endpoints, and frontend forms that integrate with your existing authentication system. + + +The assistant used the todo list because: +1. The user provided multiple complex features to implement in a comma separated list +2. The todo list helps organize these large features into manageable tasks +3. This approach allows for tracking progress across the entire implementation + + + + +User: Can you help optimize my React application? It's rendering slowly and has performance issues. +Assistant: I'll help optimize your React application. First, let me examine your codebase to identify potential performance bottlenecks. +*Reviews component structure, render patterns, state management, and data fetching* +Assistant: After analyzing your codebase, I've identified several performance issues. Let me create a todo list to track our optimization efforts. +*Creates todo list with items like: 1) Implement memoization for expensive calculations in ProductList, 2) Add virtualization for long lists in Dashboard, 3) Optimize image loading in Gallery component, 4) Fix state update loops in ShoppingCart, 5) Review bundle size and implement code splitting* +Let's start by implementing memoization for the expensive calculations in your ProductList component. + + +The assistant used the todo list because: +1. First, the assistant examined the codebase to identify specific performance issues +2. Based on this analysis, it identified multiple optimization opportunities +3. Performance optimization is a non-trivial task requiring multiple steps +4. The todo list helps methodically track improvements across different components +5. This systematic approach ensures all performance bottlenecks are addressed + + + +## Examples of When NOT to Use the Todo List + + +User: How do I print 'Hello World' in Python? +Assistant: In Python, you can print "Hello World" with this simple code: + +python +print("Hello World") + +This will output the text "Hello World" to the console when executed. + + +The assistant did not use the todo list because this is a single, trivial task that can be completed in one step. There's no need to track multiple tasks or steps for such a straightforward request. + + + + +User: What does the git status command do? +Assistant: The git status command shows the current state of your working directory and staging area. It displays which changes have been staged, which haven't, and which files aren't being tracked by Git. This is a fundamental command for understanding the current state of your repository before making commits. + + +The assistant did not use the todo list because this is an informational request with no actual coding task to complete. The user is simply asking for an explanation, not for the assistant to perform multiple steps or tasks. + + + + +User: Can you add a comment to the calculateTotal function to explain what it does? +Assistant: Sure, let me add a comment to the calculateTotal function to explain what it does. +* Uses the Edit tool to add a comment to the calculateTotal function * + + +The assistant did not use the todo list because this is a single, straightforward task confined to one location in the code. Adding a comment doesn't require tracking multiple steps or systematic organization. + + + + +User: Run npm install for me and tell me what happens. +Assistant: I'll run the npm install command for you. + +*Executes: npm install* + +The command completed successfully. Here's the output: +[Output of npm install command] + +All dependencies have been installed according to your package.json file. + + +The assistant did not use the todo list because this is a single command execution with immediate results. There are no multiple steps to track or organize, making the todo list unnecessary for this straightforward task. + + + +## Task States and Management + +1. **Task States**: Use these states to track progress: + - pending: Task not yet started + - in_progress: Currently working on (limit to ONE task at a time) + - completed: Task finished successfully + +2. **Task Management**: + - Update task status in real-time as you work + - Mark tasks complete IMMEDIATELY after finishing (don't batch completions) + - Only have ONE task in_progress at any time + - Complete current tasks before starting new ones + - Remove tasks that are no longer relevant from the list entirely + +3. **Task Completion Requirements**: + - ONLY mark a task as completed when you have FULLY accomplished it + - If you encounter errors, blockers, or cannot finish, keep the task as in_progress + - When blocked, create a new task describing what needs to be resolved + - Never mark a task as completed if: + - Tests are failing + - Implementation is partial + - You encountered unresolved errors + - You couldn't find necessary files or dependencies + +4. **Task Breakdown**: + - Create specific, actionable items + - Break complex tasks into smaller, manageable steps + - Use clear, descriptive task names + +When in doubt, use this tool. Being proactive with task management demonstrates attentiveness and ensures you complete all requirements successfully. + + +```typescript +{ + // The updated todo list + todos: { + content: string; + status: "pending" | "in_progress" | "completed"; + priority: "high" | "medium" | "low"; + id: string; + }[]; +} +``` diff --git a/tests/core/test_todo_manager.py b/tests/core/test_todo_manager.py new file mode 100644 index 0000000000..a05f5aa1d0 --- /dev/null +++ b/tests/core/test_todo_manager.py @@ -0,0 +1,108 @@ +from holmes.core.todo_manager import ( + TodoListManager, + get_todo_manager, + set_current_session_id, + get_session_id_from_context, +) +from holmes.core.tools import Task, TaskStatus + + +class TestTodoListManager: + def test_manager_creation(self): + """Test that TodoListManager can be created.""" + manager = TodoListManager() + assert manager.get_session_count() == 0 + + def test_empty_session(self): + """Test handling of empty session.""" + manager = TodoListManager() + tasks = manager.get_session_tasks("non-existent-session") + assert tasks == [] + + prompt_context = manager.format_tasks_for_prompt("non-existent-session") + assert prompt_context == "" + + def test_session_task_management(self): + """Test adding and retrieving tasks from session.""" + manager = TodoListManager() + session_id = "test-session" + + # Create test tasks + tasks = [ + Task(id="1", content="Task 1", status=TaskStatus.PENDING), + Task(id="2", content="Task 2", status=TaskStatus.IN_PROGRESS), + Task(id="3", content="Task 3", status=TaskStatus.COMPLETED), + ] + + # Update session tasks + manager.update_session_tasks(session_id, tasks) + assert manager.get_session_count() == 1 + + # Retrieve tasks + retrieved_tasks = manager.get_session_tasks(session_id) + assert len(retrieved_tasks) == 3 + assert retrieved_tasks[0].content == "Task 1" + assert retrieved_tasks[1].content == "Task 2" + assert retrieved_tasks[2].content == "Task 3" + + def test_prompt_formatting(self): + """Test that tasks are formatted correctly for prompt injection.""" + manager = TodoListManager() + session_id = "test-prompt-session" + + tasks = [ + Task(id="1", content="Check system", status=TaskStatus.PENDING), + Task(id="2", content="Review logs", status=TaskStatus.IN_PROGRESS), + Task(id="3", content="Write report", status=TaskStatus.COMPLETED), + ] + + manager.update_session_tasks(session_id, tasks) + prompt_context = manager.format_tasks_for_prompt(session_id) + + # Check structure + assert "# CURRENT INVESTIGATION TASKS" in prompt_context + assert ( + "**Task Status**: 1 completed, 1 in progress, 1 pending" in prompt_context + ) + + # Check tasks appear + assert "Check system" in prompt_context + assert "Review logs" in prompt_context + assert "Write report" in prompt_context + + # Check indicators + assert "[ ]" in prompt_context # pending + assert "[~]" in prompt_context # in_progress + assert "[✓]" in prompt_context # completed + + # Check priority + assert "(HIGH)" in prompt_context + assert "(MED)" in prompt_context + assert "(LOW)" in prompt_context + + def test_session_clearing(self): + """Test clearing session tasks.""" + manager = TodoListManager() + session_id = "test-clear-session" + + tasks = [Task(id="1", content="Task 1", status=TaskStatus.PENDING)] + manager.update_session_tasks(session_id, tasks) + assert manager.get_session_count() == 1 + + manager.clear_session(session_id) + assert manager.get_session_count() == 0 + assert manager.get_session_tasks(session_id) == [] + + def test_global_manager_instance(self): + """Test that get_todo_manager returns the same instance.""" + manager1 = get_todo_manager() + manager2 = get_todo_manager() + assert manager1 is manager2 + + def test_session_id_management(self): + """Test session ID context management.""" + test_session_id = "context-test-session" + set_current_session_id(test_session_id) + + retrieved_session_id = get_session_id_from_context() + assert retrieved_session_id == test_session_id diff --git a/tests/core/test_todo_write_tool.py b/tests/core/test_todo_write_tool.py new file mode 100644 index 0000000000..8e3b8c75cf --- /dev/null +++ b/tests/core/test_todo_write_tool.py @@ -0,0 +1,183 @@ +from holmes.core.tools import TodoWriteTool, ToolResultStatus, TaskStatus, TaskPriority + + +class TestTodoWriteTool: + def test_todo_write_tool_creation(self): + """Test that TodoWriteTool can be created with correct parameters.""" + tool = TodoWriteTool() + assert tool.name == "TodoWrite" + assert "investigation tasks" in tool.description + assert "todos" in tool.parameters + + def test_todo_write_tool_empty_params(self): + """Test TodoWriteTool with empty parameters.""" + tool = TodoWriteTool() + result = tool._invoke({}) + + assert result.status == ToolResultStatus.SUCCESS + assert isinstance(result.data, str) + assert "0 tasks" in result.data + assert "Investigation plan updated" in result.data + + def test_todo_write_tool_with_tasks(self): + """Test TodoWriteTool with valid task data.""" + tool = TodoWriteTool() + params = { + "todos": [ + { + "id": "1", + "content": "Check pod status", + "status": "pending", + "priority": "high", + }, + { + "id": "2", + "content": "Analyze logs", + "status": "in_progress", + "priority": "medium", + }, + ] + } + + result = tool._invoke(params) + + assert result.status == ToolResultStatus.SUCCESS + assert isinstance(result.data, str) + assert "2 tasks" in result.data + assert "Investigation plan updated" in result.data + # Should include pretty printed TodoList + assert "Check pod status" in result.data + assert "Analyze logs" in result.data + + def test_todo_write_tool_default_values(self): + """Test TodoWriteTool with minimal task data uses defaults.""" + tool = TodoWriteTool() + params = {"todos": [{"content": "Test task"}]} + + result = tool._invoke(params) + + assert result.status == ToolResultStatus.SUCCESS + assert isinstance(result.data, str) + assert "1 tasks" in result.data + assert "Investigation plan updated" in result.data + # Should include pretty printed TodoList + assert "Test task" in result.data + + def test_todo_write_tool_invalid_enum_values(self): + """Test TodoWriteTool handles invalid enum values gracefully.""" + tool = TodoWriteTool() + params = { + "todos": [ + { + "content": "Test task", + "status": "invalid_status", + "priority": "invalid_priority", + } + ] + } + + result = tool._invoke(params) + + # Should handle gracefully and return error + assert result.status == ToolResultStatus.ERROR + assert "Failed to process tasks" in result.error + + def test_get_parameterized_one_liner(self): + """Test the parameterized one-liner description.""" + tool = TodoWriteTool() + + params = {"todos": [{"content": "task1"}, {"content": "task2"}]} + one_liner = tool.get_parameterized_one_liner(params) + assert "2 investigation tasks" in one_liner + + params = {"todos": []} + one_liner = tool.get_parameterized_one_liner(params) + assert "0 investigation tasks" in one_liner + + def test_task_status_enum(self): + """Test TaskStatus enum values.""" + assert TaskStatus.PENDING == "pending" + assert TaskStatus.IN_PROGRESS == "in_progress" + assert TaskStatus.COMPLETED == "completed" + + def test_task_priority_enum(self): + """Test TaskPriority enum values.""" + assert TaskPriority.HIGH == "high" + assert TaskPriority.MEDIUM == "medium" + assert TaskPriority.LOW == "low" + + def test_openai_format(self): + """Test that the tool generates correct OpenAI format.""" + tool = TodoWriteTool() + openai_format = tool.get_openai_format() + + assert openai_format["type"] == "function" + assert openai_format["function"]["name"] == "TodoWrite" + assert "investigation tasks" in openai_format["function"]["description"] + + # Check parameters schema + params = openai_format["function"]["parameters"] + assert params["type"] == "object" + assert "todos" in params["properties"] + + # Check array schema has items property + todos_param = params["properties"]["todos"] + assert todos_param["type"] == "array" + assert "items" in todos_param + assert todos_param["items"]["type"] == "object" + + # Check required fields + assert "todos" in params["required"] + + def test_session_storage_functionality(self): + """Test that the tool stores tasks in session storage.""" + from holmes.core.todo_manager import get_todo_manager, set_current_session_id + + tool = TodoWriteTool() + session_id = "test-session-456" + set_current_session_id(session_id) + + params = { + "todos": [ + { + "id": "1", + "content": "Task 1", + "status": "pending", + "priority": "high", + }, + { + "id": "2", + "content": "Task 2", + "status": "completed", + "priority": "low", + }, + ] + } + + result = tool._invoke(params) + + assert result.status == ToolResultStatus.SUCCESS + assert "2 tasks" in result.data + assert "Investigation plan updated" in result.data + + # Check that the pretty printed TodoList is included in the response + assert "CURRENT INVESTIGATION TASKS" in result.data + assert "Task 1" in result.data + assert "Task 2" in result.data + assert "[ ]" in result.data # pending indicator + assert "[✓]" in result.data # completed indicator + assert "(HIGH)" in result.data # priority indicator + assert "(LOW)" in result.data # priority indicator + + # Check session storage + manager = get_todo_manager() + stored_tasks = manager.get_session_tasks(session_id) + assert len(stored_tasks) == 2 + assert stored_tasks[0].content == "Task 1" + assert stored_tasks[1].content == "Task 2" + + # Check prompt formatting matches what's in the response + prompt_context = manager.format_tasks_for_prompt(session_id) + assert ( + prompt_context in result.data + ) # The formatted tasks should be part of the response diff --git a/tests/plugins/toolsets/test_core_investigation.py b/tests/plugins/toolsets/test_core_investigation.py new file mode 100644 index 0000000000..361c96a616 --- /dev/null +++ b/tests/plugins/toolsets/test_core_investigation.py @@ -0,0 +1,41 @@ +from holmes.plugins.toolsets.investigator.core_investigation import ( + CoreInvestigationToolset, +) +from holmes.core.tools import ToolsetStatusEnum, ToolsetTag, TodoWriteTool + + +class TestCoreInvestigationToolset: + def test_toolset_creation(self): + """Test that CoreInvestigationToolset is created correctly.""" + toolset = CoreInvestigationToolset() + + assert toolset.name == "core_investigation" + assert "investigation tools" in toolset.description + assert toolset.enabled is True + assert toolset.is_default is True + assert ToolsetTag.CORE in toolset.tags + + def test_toolset_has_todo_write_tool(self): + """Test that the toolset includes the TodoWrite tool.""" + toolset = CoreInvestigationToolset() + + assert len(toolset.tools) == 1 + assert isinstance(toolset.tools[0], TodoWriteTool) + assert toolset.tools[0].name == "TodoWrite" + + def test_toolset_check_prerequisites(self): + """Test that toolset prerequisites check passes.""" + toolset = CoreInvestigationToolset() + toolset.check_prerequisites() + + # Should be enabled by default with no prerequisites + assert toolset.status == ToolsetStatusEnum.ENABLED + assert toolset.error is None + + def test_get_example_config(self): + """Test that example config is returned.""" + toolset = CoreInvestigationToolset() + config = toolset.get_example_config() + + assert isinstance(config, dict) + # Core toolset doesn't need configuration From d0dc615d644ac1969ed513b5f433107f8f55c55d Mon Sep 17 00:00:00 2001 From: Arik Alon Date: Tue, 12 Aug 2025 15:12:23 +0300 Subject: [PATCH 02/17] improvements --- holmes/core/todo_manager.py | 3 +- holmes/core/tools.py | 23 ++++++------- holmes/plugins/prompts/generic_ask.jinja2 | 33 ++++++++++++++++++- .../test_case.yaml | 9 +++++ .../toolsets.yaml | 3 ++ 5 files changed, 56 insertions(+), 15 deletions(-) create mode 100644 tests/llm/fixtures/test_ask_holmes/150_what_is_wrong_with_mimir/test_case.yaml create mode 100644 tests/llm/fixtures/test_ask_holmes/150_what_is_wrong_with_mimir/toolsets.yaml diff --git a/holmes/core/todo_manager.py b/holmes/core/todo_manager.py index b80722406c..ba661fc0a1 100644 --- a/holmes/core/todo_manager.py +++ b/holmes/core/todo_manager.py @@ -108,7 +108,8 @@ def get_session_id_from_context(context=None) -> str: # Generate new session ID session_id = str(uuid4()) - context._current_session = session_id + if context: + context._current_session = session_id return session_id diff --git a/holmes/core/tools.py b/holmes/core/tools.py index 344cd47e2f..f99b86ca3b 100644 --- a/holmes/core/tools.py +++ b/holmes/core/tools.py @@ -602,12 +602,13 @@ class TodoWriteTool(Tool): } def _invoke(self, params: Dict) -> StructuredToolResult: - try: - from holmes.core.todo_manager import ( - get_todo_manager, - get_session_id_from_context, - ) + from holmes.core.todo_manager import ( + get_todo_manager, + get_session_id_from_context, + ) + try: + logging.info(f"### invoking todowrite tool {params}") todos_data = params.get("todos", []) tasks = [] @@ -638,15 +639,11 @@ def print_tasks_table(tasks): max_id_width = max(len(str(task.id)) for task in tasks) max_content_width = max(len(task.content) for task in tasks) max_status_width = max(len(task.status.value) for task in tasks) - max_priority_width = max( - len(task.priority.value.upper()) for task in tasks - ) # Ensure minimum widths for headers id_width = max(max_id_width, 2) content_width = max(max_content_width, 7) status_width = max(max_status_width, 6) - priority_width = max(max_priority_width, 8) # Status indicators status_icons = { @@ -656,8 +653,8 @@ def print_tasks_table(tasks): } # Build table - separator = f"+{'-' * (id_width + 2)}+{'-' * (content_width + 2)}+{'-' * (status_width + 2)}+{'-' * (priority_width + 2)}+" - header = f"| {'ID':<{id_width}} | {'Content':<{content_width}} | {'Status':<{status_width}} | {'Priority':<{priority_width}} |" + separator = f"+{'-' * (id_width + 2)}+{'-' * (content_width + 2)}+{'-' * (status_width + 2)}+" + header = f"| {'ID':<{id_width}} | {'Content':<{content_width}} | {'Status':<{status_width}} |" # Log the table logging.info("Updated Investigation Tasks:") @@ -669,8 +666,7 @@ def print_tasks_table(tasks): status_display = ( f"{status_icons[task.status.value]} {task.status.value}" ) - priority_display = task.priority.value.upper() - row = f"| {task.id:<{id_width}} | {task.content:<{content_width}} | {status_display:<{status_width}} | {priority_display:<{priority_width}} |" + row = f"| {task.id:<{id_width}} | {task.content:<{content_width}} | {status_display:<{status_width}} |" logging.info(row) logging.info(separator) @@ -695,6 +691,7 @@ def print_tasks_table(tasks): ) except Exception as e: + logging.exception("error using todowrite tool") return StructuredToolResult( status=ToolResultStatus.ERROR, error=f"Failed to process tasks: {str(e)}", diff --git a/holmes/plugins/prompts/generic_ask.jinja2 b/holmes/plugins/prompts/generic_ask.jinja2 index 9103635824..301aca1024 100644 --- a/holmes/plugins/prompts/generic_ask.jinja2 +++ b/holmes/plugins/prompts/generic_ask.jinja2 @@ -5,7 +5,38 @@ Do not say 'based on the tool output' or explicitly refer to tools at all. If you output an answer and then realize you need to call more tools or there are possible next steps, you may do so by calling tools at that point in time. If you have a good and concrete suggestion for how the user can fix something, tell them even if not asked explicitly -CRITICAL: For multi-step questions, you MUST start by calling the TodoWrite tool to create your investigation plan. +CRITICAL: For multi-step questions, you MUST start by calling the TodoWrite tool with a `todos` parameter containing an array of task objects. Each task must have: +- `id`: unique identifier (string) +- `content`: specific task description (string) +- `status`: "pending" for new tasks (string) + +MANDATORY Task Status Updates: +- When starting a task: Call TodoWrite changing that task's status to "in_progress" +- When completing a task: Call TodoWrite changing that task's status to "completed" +- Keep all other tasks unchanged in each TodoWrite call +- Only work on ONE task at a time (only one "in_progress" task) + +# MANDATORY Self-Verification Checklist + +Before providing any final answer, you MUST create and complete a verification task using TodoWrite that checks: +✓ **Completeness**: Does my answer fully address what the user asked? +✓ **Specificity**: Have I provided exact resource names, namespaces, versions, and commands? +✓ **Root Cause**: Did I use "five whys" to find the actual root cause, not just symptoms? +✓ **Actionability**: Have I given concrete next steps the user can take? +✓ **Evidence**: Is every claim in my answer backed by tool output or investigation data? +✓ **Assumptions**: Have I verified all resource names, namespaces, and configurations through tools rather than assuming? + +Only after completing this verification checklist should you provide your final answer. + + Status Update Example: + # Starting task 2: + TodoWrite(todos=[ + {"id": "1", "content": "Check pod status", "status": "completed"}, + {"id": "2", "content": "Examine logs", "status": "in_progress"}, + {"id": "3", "content": "Check resources", "status": "pending"} + {"id": "4", "content": "Investigation Verification", "status": "pending"} + ]) + {% if todo_list %} {{ todo_list }} diff --git a/tests/llm/fixtures/test_ask_holmes/150_what_is_wrong_with_mimir/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/150_what_is_wrong_with_mimir/test_case.yaml new file mode 100644 index 0000000000..621ab86687 --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/150_what_is_wrong_with_mimir/test_case.yaml @@ -0,0 +1,9 @@ +user_prompt: + - "What is wrong with mimir?" +expected_output: | + Output must contain 1 or more embeds in the following format + <<{"type": "datadogql", "tool_name": "query_datadog_metrics", "random_key": "iD8G"}>> + + random_key may be different than the above example, but all other parameters (type and tool_name) must be as described + + Output must NOT tell the user it doesn't have access to metrics or that they should use another tool diff --git a/tests/llm/fixtures/test_ask_holmes/150_what_is_wrong_with_mimir/toolsets.yaml b/tests/llm/fixtures/test_ask_holmes/150_what_is_wrong_with_mimir/toolsets.yaml new file mode 100644 index 0000000000..3c9a2dc247 --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/150_what_is_wrong_with_mimir/toolsets.yaml @@ -0,0 +1,3 @@ +toolsets: + kubernetes/core: + enabled: true From 0e280904593b3fd6b50a49261441e0ebc0fa36c2 Mon Sep 17 00:00:00 2001 From: Arik Alon Date: Wed, 6 Aug 2025 20:25:03 +0300 Subject: [PATCH 03/17] subtasks commit --- holmes/core/investigation.py | 8 + holmes/core/prompt.py | 7 + holmes/core/todo_manager.py | 118 ++++++++ holmes/core/tool_calling_llm.py | 11 + holmes/core/tools.py | 135 +++++++++ .../prompts/_general_instructions.jinja2 | 12 + holmes/plugins/prompts/generic_ask.jinja2 | 24 ++ .../prompts/generic_investigation.jinja2 | 28 ++ holmes/plugins/toolsets/__init__.py | 4 + .../plugins/toolsets/investigator/__init__.py | 0 .../investigator/core_investigation.py | 32 ++ .../investigator_instructions.jinja2 | 273 ++++++++++++++++++ tests/core/test_todo_manager.py | 108 +++++++ tests/core/test_todo_write_tool.py | 183 ++++++++++++ .../toolsets/test_core_investigation.py | 41 +++ 15 files changed, 984 insertions(+) create mode 100644 holmes/core/todo_manager.py create mode 100644 holmes/plugins/toolsets/investigator/__init__.py create mode 100644 holmes/plugins/toolsets/investigator/core_investigation.py create mode 100644 holmes/plugins/toolsets/investigator/investigator_instructions.jinja2 create mode 100644 tests/core/test_todo_manager.py create mode 100644 tests/core/test_todo_write_tool.py create mode 100644 tests/plugins/toolsets/test_core_investigation.py diff --git a/holmes/core/investigation.py b/holmes/core/investigation.py index b8cd503e68..918e01537d 100644 --- a/holmes/core/investigation.py +++ b/holmes/core/investigation.py @@ -133,6 +133,13 @@ def get_investigation_context( else: logging.info("Structured output is disabled for this request") + # Add TodoList context to prompt + from holmes.core.todo_manager import get_todo_manager, get_session_id_from_context + + todo_manager = get_todo_manager() + session_id = get_session_id_from_context() + todo_context = todo_manager.format_tasks_for_prompt(session_id) + system_prompt = load_and_render_prompt( investigate_request.prompt_template, { @@ -141,6 +148,7 @@ def get_investigation_context( "structured_output": request_structured_output_from_llm, "toolsets": ai.tool_executor.toolsets, "cluster_name": config.cluster_name, + "todo_list": todo_context, }, ) diff --git a/holmes/core/prompt.py b/holmes/core/prompt.py index e2e8ef4fd0..c317658e8d 100644 --- a/holmes/core/prompt.py +++ b/holmes/core/prompt.py @@ -59,6 +59,13 @@ def build_initial_ask_messages( console, initial_user_prompt, file_paths ) + user_prompt_with_files += ( + "\n\n\nIMPORTANT: You have access to the TodoWrite tool. It creates a TodoList, in order to track progress. It's very important. You MUST use it:\n1. FIRST: Ask your self which sub problems you need to solve in order to answer the question." + "Do this, BEFORE any other tools\n2. " + "AFTER EVERY TOOL CALL: If required, update the TodoList\n3. " + "\n\nFAILURE TO UPDATE TodoList = INCOMPLETE INVESTIGATION\n\n" + "Example flow:\n- Think and divide to sub problems → create TodoList → Perform each task on the list → Update list → Verify your solution\n" + ) messages = [ {"role": "system", "content": system_prompt_rendered}, {"role": "user", "content": user_prompt_with_files}, diff --git a/holmes/core/todo_manager.py b/holmes/core/todo_manager.py new file mode 100644 index 0000000000..b80722406c --- /dev/null +++ b/holmes/core/todo_manager.py @@ -0,0 +1,118 @@ +import logging +from typing import Dict, List +from threading import Lock +from uuid import uuid4 + +from holmes.core.tools import Task, TaskStatus + + +class TodoListManager: + """ + Session-based storage manager for investigation TodoLists. + Stores TodoLists per session and provides methods to get/update tasks. + """ + + def __init__(self): + self._sessions: Dict[str, List[Task]] = {} + self._lock = Lock() + + def get_session_tasks(self, session_id: str) -> List[Task]: + """Get all tasks for a session. Returns empty list if session doesn't exist.""" + logging.info(f"########## get_session_tasks {session_id}") + with self._lock: + return self._sessions.get(session_id, []).copy() + + def update_session_tasks(self, session_id: str, tasks: List[Task]) -> None: + """Update all tasks for a session.""" + with self._lock: + self._sessions[session_id] = tasks.copy() + + def clear_session(self, session_id: str) -> None: + """Clear all tasks for a session.""" + with self._lock: + if session_id in self._sessions: + del self._sessions[session_id] + + def get_session_count(self) -> int: + """Get number of active sessions (for debugging/testing).""" + with self._lock: + return len(self._sessions) + + def format_tasks_for_prompt(self, session_id: str) -> str: + """ + Format tasks for injection into system prompt. + Returns empty string if no tasks exist. + """ + tasks = self.get_session_tasks(session_id) + + if not tasks: + return "" + + # Sort tasks by status (pending -> in_progress -> completed) then priority + status_order = {"pending": 0, "in_progress": 1, "completed": 2} + + sorted_tasks = sorted( + tasks, + key=lambda t: (status_order.get(t.status.value, 3),), + ) + + lines = ["# CURRENT INVESTIGATION TASKS"] + lines.append("") + + # Count tasks by status + pending_count = sum(1 for t in tasks if t.status == TaskStatus.PENDING) + progress_count = sum(1 for t in tasks if t.status == TaskStatus.IN_PROGRESS) + completed_count = sum(1 for t in tasks if t.status == TaskStatus.COMPLETED) + + lines.append( + f"**Task Status**: {completed_count} completed, {progress_count} in progress, {pending_count} pending" + ) + lines.append("") + + for task in sorted_tasks: + # Use simple text indicators for prompt injection + status_indicator = { + "pending": "[ ]", + "in_progress": "[~]", + "completed": "[✓]", + }.get(task.status.value, "[?]") + + lines.append(f"{status_indicator} [{task.id}] {task.content}") + + lines.append("") + lines.append( + "**Instructions**: Use TodoWrite tool to update task status as you work. Mark tasks as 'in_progress' when starting, 'completed' when finished." + ) + + return "\n".join(lines) + + +# Global instance for session management +_todo_manager = TodoListManager() + + +def get_todo_manager() -> TodoListManager: + """Get the global TodoListManager instance.""" + return _todo_manager + + +def get_session_id_from_context(context=None) -> str: + """ + Extract or generate session ID from context. + For now, we'll use a simple approach - this can be enhanced later. + """ + # TODO: This should be enhanced to extract session ID from investigation context + # For now, we'll use a simple thread-local or context-based approach + if hasattr(context, "_current_session"): + return context._current_session + + # Generate new session ID + session_id = str(uuid4()) + context._current_session = session_id + return session_id + + +def set_current_session_id(session_id: str) -> None: + """Set the current session ID for this thread/context.""" + pass + # get_session_id_from_context._current_session = session_id diff --git a/holmes/core/tool_calling_llm.py b/holmes/core/tool_calling_llm.py index e6b9baa10d..9726491cf6 100644 --- a/holmes/core/tool_calling_llm.py +++ b/holmes/core/tool_calling_llm.py @@ -776,6 +776,16 @@ def investigate( "[bold]No runbooks found for this issue. Using default behaviour. (Add runbooks to guide the investigation.)[/bold]" ) + # Add TodoList context to prompt + from holmes.core.todo_manager import ( + get_todo_manager, + get_session_id_from_context, + ) + + todo_manager = get_todo_manager() + session_id = get_session_id_from_context() + todo_context = todo_manager.format_tasks_for_prompt(session_id) + system_prompt = load_and_render_prompt( prompt, { @@ -784,6 +794,7 @@ def investigate( "structured_output": request_structured_output_from_llm, "toolsets": self.tool_executor.toolsets, "cluster_name": self.cluster_name, + "todo_list": todo_context, }, ) diff --git a/holmes/core/tools.py b/holmes/core/tools.py index cff65e18f0..344cd47e2f 100644 --- a/holmes/core/tools.py +++ b/holmes/core/tools.py @@ -18,6 +18,7 @@ from holmes.plugins.prompts import load_and_render_prompt import time from rich.table import Table +from uuid import uuid4 class ToolResultStatus(str, Enum): @@ -569,3 +570,137 @@ def pretty_print_toolset_status(toolsets: list[Toolset], console: Console) -> No table.add_row(*(str(row.get(col.capitalize(), "")) for col in status_fields)) console.print(table) + + +class TaskStatus(str, Enum): + PENDING = "pending" + IN_PROGRESS = "in_progress" + COMPLETED = "completed" + + +class TaskPriority(str, Enum): + HIGH = "high" + MEDIUM = "medium" + LOW = "low" + + +class Task(BaseModel): + id: str = Field(default_factory=lambda: str(uuid4())) + content: str + status: TaskStatus = TaskStatus.PENDING + + +class TodoWriteTool(Tool): + name: str = "TodoWrite" + description: str = "Save investigation tasks to break down complex problems into manageable sub-tasks" + parameters: Dict[str, ToolParameter] = { + "todos": ToolParameter( + description="List of tasks to track during the investigation. Each task should have: id (string), content (string), status (pending/in_progress/completed)", + type="array[object]", + required=True, + ) + } + + def _invoke(self, params: Dict) -> StructuredToolResult: + try: + from holmes.core.todo_manager import ( + get_todo_manager, + get_session_id_from_context, + ) + + todos_data = params.get("todos", []) + + tasks = [] + + for todo_item in todos_data: + if isinstance(todo_item, dict): + task = Task( + id=todo_item.get("id", str(uuid4())), + content=todo_item.get("content", ""), + status=TaskStatus(todo_item.get("status", "pending")), + ) + tasks.append(task) + + logging.info(f"Tasks: {len(tasks)}") + + # Store tasks in session storage + todo_manager = get_todo_manager() + session_id = get_session_id_from_context() + todo_manager.update_session_tasks(session_id, tasks) + + # Print a nice table to console/log + def print_tasks_table(tasks): + if not tasks: + logging.info("No tasks in the investigation plan.") + return + + # Calculate column widths + max_id_width = max(len(str(task.id)) for task in tasks) + max_content_width = max(len(task.content) for task in tasks) + max_status_width = max(len(task.status.value) for task in tasks) + max_priority_width = max( + len(task.priority.value.upper()) for task in tasks + ) + + # Ensure minimum widths for headers + id_width = max(max_id_width, 2) + content_width = max(max_content_width, 7) + status_width = max(max_status_width, 6) + priority_width = max(max_priority_width, 8) + + # Status indicators + status_icons = { + "pending": "[ ]", + "in_progress": "[~]", + "completed": "[✓]", + } + + # Build table + separator = f"+{'-' * (id_width + 2)}+{'-' * (content_width + 2)}+{'-' * (status_width + 2)}+{'-' * (priority_width + 2)}+" + header = f"| {'ID':<{id_width}} | {'Content':<{content_width}} | {'Status':<{status_width}} | {'Priority':<{priority_width}} |" + + # Log the table + logging.info("Updated Investigation Tasks:") + logging.info(separator) + logging.info(header) + logging.info(separator) + + for task in tasks: + status_display = ( + f"{status_icons[task.status.value]} {task.status.value}" + ) + priority_display = task.priority.value.upper() + row = f"| {task.id:<{id_width}} | {task.content:<{content_width}} | {status_display:<{status_width}} | {priority_display:<{priority_width}} |" + logging.info(row) + + logging.info(separator) + + # Print table to console/log + print_tasks_table(tasks) + + # Get pretty formatted version of the updated TodoList (keep existing format) + formatted_tasks = todo_manager.format_tasks_for_prompt(session_id) + + # Return confirmation with pretty printed TodoList (unchanged) + response_data = f"✅ Investigation plan updated with {len(tasks)} tasks. Tasks are now stored in session and will appear in subsequent prompts.\n\n" + if formatted_tasks: + response_data += formatted_tasks + else: + response_data += "No tasks currently in the investigation plan." + + return StructuredToolResult( + status=ToolResultStatus.SUCCESS, + data=response_data, + params=params, + ) + + except Exception as e: + return StructuredToolResult( + status=ToolResultStatus.ERROR, + error=f"Failed to process tasks: {str(e)}", + params=params, + ) + + def get_parameterized_one_liner(self, params: Dict) -> str: + todos = params.get("todos", []) + return f"Write {todos} investigation tasks" diff --git a/holmes/plugins/prompts/_general_instructions.jinja2 b/holmes/plugins/prompts/_general_instructions.jinja2 index b9e5941b6a..a8e5b8d053 100644 --- a/holmes/plugins/prompts/_general_instructions.jinja2 +++ b/holmes/plugins/prompts/_general_instructions.jinja2 @@ -47,6 +47,18 @@ * For any question, try to make the answer specific to the user's cluster. ** For example, if asked to port forward, find out the app or pod port (kubectl describe) and provide a port forward command specific to the user's question +# MANDATORY Task Management + +* You MUST use the TodoWrite tool for ANY investigation requiring multiple steps +* Your FIRST tool call MUST be TodoWrite to create your investigation plan +* Break down ALL complex problems into smaller, manageable tasks +* You MUST update task status (pending → in_progress → completed) as you work through your investigation +* The TodoWrite tool will show you a formatted task list - reference this throughout your investigation +* Mark tasks as 'in_progress' when you start them, 'completed' when finished +* Follow ALL tasks in your plan - don't skip any tasks +* Use task management to ensure you don't miss important investigation steps +* If you discover additional steps during investigation, add them to your task list using TodoWrite + # Tool/function calls You are able to make tool calls / function calls. Recognise when a tool has already been called and reuse its result. diff --git a/holmes/plugins/prompts/generic_ask.jinja2 b/holmes/plugins/prompts/generic_ask.jinja2 index fb7d9b35c8..9103635824 100644 --- a/holmes/plugins/prompts/generic_ask.jinja2 +++ b/holmes/plugins/prompts/generic_ask.jinja2 @@ -4,6 +4,30 @@ Ask for multiple tool calls at the same time as it saves time for the user. Do not say 'based on the tool output' or explicitly refer to tools at all. If you output an answer and then realize you need to call more tools or there are possible next steps, you may do so by calling tools at that point in time. If you have a good and concrete suggestion for how the user can fix something, tell them even if not asked explicitly + +CRITICAL: For multi-step questions, you MUST start by calling the TodoWrite tool to create your investigation plan. + +{% if todo_list %} +{{ todo_list }} +{% endif %} + +# MANDATORY Investigation Process + +For ANY question that requires multiple steps or investigation, you MUST follow this structured approach: + +1. **IMMEDIATELY START with TodoWrite**: Before doing any investigation, you MUST call the TodoWrite tool to create an investigation plan. Break down the question into specific tasks. + +2. **Problem Analysis**: Identify what sub-problems need to be solved and create tasks for each. + +3. **Task Execution**: Execute your plan systematically. You MUST: + - Mark tasks as 'in_progress' when you start working on them + - Call TodoWrite to update task status as you complete each task + - Reference the visible task list from your TodoWrite output throughout your investigation + - Follow the exact tasks you created - complete all of them + +4. **Verification**: Before providing your final answer, verify that your conclusion is complete and addresses the user's question fully. + +IMPORTANT: If a question requires more than one investigation step, your FIRST tool call MUST be TodoWrite. {% include '_current_date_time.jinja2' %} Use conversation history to maintain continuity when appropriate, ensuring efficiency in your responses. diff --git a/holmes/plugins/prompts/generic_investigation.jinja2 b/holmes/plugins/prompts/generic_investigation.jinja2 index 2789932afe..d0ac1195d5 100644 --- a/holmes/plugins/prompts/generic_investigation.jinja2 +++ b/holmes/plugins/prompts/generic_investigation.jinja2 @@ -3,6 +3,34 @@ Whenever possible you MUST first use tools to investigate then answer the questi Ask for multiple tool calls at the same time as it saves time for the user. Do not say 'based on the tool output' +CRITICAL: For ALL investigations, you MUST start by calling the TodoWrite tool to create your investigation plan. + +{% if todo_list %} +{{ todo_list }} +{% endif %} + +# MANDATORY Investigation Process + +You MUST follow this structured approach for ALL investigations. NO EXCEPTIONS: + +1. **IMMEDIATELY START with TodoWrite**: Before doing ANYTHING else, you MUST call the TodoWrite tool to create an investigation plan. Analyze the issue and break it down into specific tasks. + +2. **Problem Analysis**: Ask yourself: "What are the sub-problems I need to solve to answer this question?" Create tasks for each sub-problem. + +3. **Task Execution**: Execute your investigation plan by calling the appropriate tools. You MUST: + - Mark tasks as 'in_progress' when you start working on them + - Call TodoWrite to update task status as you complete each task + - Reference the visible task list from your TodoWrite output throughout your investigation + - Follow the exact tasks you created - don't skip or ignore any + +4. **Verification**: Once you believe you have found the answer, perform a verification step: + - Cross-check your findings with the original alert/issue details + - Ensure your conclusion addresses the root cause, not just symptoms + - Verify that your analysis is consistent with all collected data + - If verification fails, use TodoWrite to add new tasks and return to investigation + +REMEMBER: Your first tool call MUST ALWAYS be TodoWrite to create your investigation plan. + Provide an terse analysis of the following {{ issue.source_type }} alert/issue and why it is firing. * {% include '_current_date_time.jinja2' %} * If the tool requires string format timestamps, query from 'start_timestamp' until 'end_timestamp' diff --git a/holmes/plugins/toolsets/__init__.py b/holmes/plugins/toolsets/__init__.py index 6932c82edf..50c4687f38 100644 --- a/holmes/plugins/toolsets/__init__.py +++ b/holmes/plugins/toolsets/__init__.py @@ -44,6 +44,9 @@ from holmes.plugins.toolsets.robusta.robusta import RobustaToolset from holmes.plugins.toolsets.runbook.runbook_fetcher import RunbookToolset from holmes.plugins.toolsets.servicenow.servicenow import ServiceNowToolset +from holmes.plugins.toolsets.investigator.core_investigation import ( + CoreInvestigationToolset, +) THIS_DIR = os.path.abspath(os.path.dirname(__file__)) @@ -68,6 +71,7 @@ def load_toolsets_from_file( def load_python_toolsets(dal: Optional[SupabaseDal]) -> List[Toolset]: logging.debug("loading python toolsets") toolsets: list[Toolset] = [ + CoreInvestigationToolset(), # Load first for higher priority InternetToolset(), RobustaToolset(dal), OpenSearchToolset(), diff --git a/holmes/plugins/toolsets/investigator/__init__.py b/holmes/plugins/toolsets/investigator/__init__.py new file mode 100644 index 0000000000..e69de29bb2 diff --git a/holmes/plugins/toolsets/investigator/core_investigation.py b/holmes/plugins/toolsets/investigator/core_investigation.py new file mode 100644 index 0000000000..c3e927f99b --- /dev/null +++ b/holmes/plugins/toolsets/investigator/core_investigation.py @@ -0,0 +1,32 @@ +import logging +import os +from typing import Any, Dict + +from holmes.core.tools import TodoWriteTool, Toolset, ToolsetTag, ToolsetStatusEnum + + +class CoreInvestigationToolset(Toolset): + """Core toolset for investigation management and task planning.""" + + def __init__(self): + super().__init__( + name="core_investigation", + description="Core investigation tools for task management and planning", + enabled=True, + tools=[TodoWriteTool()], + tags=[ToolsetTag.CORE], + is_default=True, + ) + # Override the default DISABLED status to ENABLED since this is a core toolset + self.status = ToolsetStatusEnum.ENABLED + logging.info("Core investigation toolset loaded") + + def get_example_config(self) -> Dict[str, Any]: + return {} + + def _reload_instructions(self): + """Load Datadog metrics specific troubleshooting instructions.""" + template_file_path = os.path.abspath( + os.path.join(os.path.dirname(__file__), "investigator_instructions.jinja2") + ) + self._load_llm_instructions(jinja_template=f"file://{template_file_path}") diff --git a/holmes/plugins/toolsets/investigator/investigator_instructions.jinja2 b/holmes/plugins/toolsets/investigator/investigator_instructions.jinja2 new file mode 100644 index 0000000000..c8f6da4b10 --- /dev/null +++ b/holmes/plugins/toolsets/investigator/investigator_instructions.jinja2 @@ -0,0 +1,273 @@ +# Task Management +You have access to the TodoWrite tool to help you manage and plan tasks. The updated task list will always appear in the prompt. Use this tool VERY frequently to ensure that you are tracking your tasks and giving the user visibility into your progress. +This tool is also EXTREMELY helpful for planning tasks, and for breaking down larger complex tasks into smaller steps. If you do not use this tool when planning, you may forget to do important tasks - and that is unacceptable. + +It is critical that you mark todos as completed as soon as you are done with a task. Do not batch up multiple tasks before marking them as completed. + +Examples: + + +user: Run the build and fix any type errors +assistant: I'm going to use the TodoWrite tool to write the following items to the todo list: +- Run the build +- Fix any type errors + +I'm now going to run the build using Bash. + +Looks like I found 10 type errors. I'm going to use the TodoWrite tool to write 10 items to the todo list. + +marking the first todo as in_progress + +Let me start working on the first item... + +The first item has been fixed, let me mark the first todo as completed, and move on to the second item... +.. +.. + +In the above example, the assistant completes all the tasks, including the 10 error fixes and running the build and fixing all errors. + + +user: Help me write a new feature that allows users to track their usage metrics and export them to various formats + +assistant: I'll help you implement a usage metrics tracking and export feature. Let me first use the TodoWrite tool to plan this task. +Adding the following todos to the todo list: +1. Research existing metrics tracking in the codebase +2. Design the metrics collection system +3. Implement core metrics tracking functionality +4. Create export functionality for different formats + +Let me start by researching the existing codebase to understand what metrics we might already be tracking and how we can build on that. + +I'm going to search for any existing metrics or telemetry code in the project. + +I've found some existing telemetry code. Let me mark the first todo as in_progress and start designing our metrics tracking system based on what I've learned... + +[Assistant continues implementing the feature step by step, marking todos as in_progress and completed as they go] + + + +# Doing tasks +The user will primarily request you perform reliability troubleshooting tasks. This includes finding bugs, finding misconfigurations, looking at observability data, including logs, traces, metrics, following investigation runbooks, searching for changes that might be related to incidents, adding new functionality, explaining code, and more. For these tasks the following steps are recommended: +- Use the TodoWrite tool to plan the task if required +- Use the available tools to understand the codebase, environment and the user's query. You are encouraged to use the tools extensively both in parallel and sequentially. +- Implement the solution using all tools available to you +- Verify the solution if possible. +- VERY IMPORTANT: When you have completed a task, validate that the answer you provided is correct. + +- Tool results and user messages may include tags. tags contain useful information and reminders. They are NOT part of the user's provided input or the tool result. + +# Tool usage policy +- You have the capability to call multiple tools in a single response. When multiple independent pieces of information are requested, batch your tool calls together for optimal performance. When making multiple bash tool calls, you MUST send a single message with multiple tools calls to run the calls in parallel. For example, if you need to run "git status" and "git diff", send a single message with two tool calls to run the calls in parallel. + +You MUST answer concisely with fewer than 4 lines of text (not including tool use or code generation), unless user asks for detail. + + +Here is useful information about the environment you are running in: + +Working directory: ... +Is directory a git repo: Yes +Platform: macos +OS Version: Darwin 24.1.0 +Today's date: 2025/6/13 + +You are powered by the model named Sonnet 4. The exact model ID is claude-sonnet-4-20250514. + + +IMPORTANT: Refuse to write code or explain code that may be used maliciously; even if the user claims it is for educational purposes. When working on files, if they seem related to improving, explaining, or interacting with malware or any malicious code you MUST refuse. +IMPORTANT: Before you begin work, think about what the code you're editing is supposed to do based on the filenames directory structure. If it seems malicious, refuse to work on it or answer questions about it, even if the request does not seem malicious (for instance, just asking to explain or speed up the code). + + +IMPORTANT: Always use the TodoWrite tool to plan and track tasks throughout the conversation. + + + + +# TodoWrite +Use this tool to create and manage a structured task list for your current coding session. This helps you track progress, organize complex tasks, and demonstrate thoroughness to the user. +It also helps the user understand the progress of the task and overall progress of their requests. + +## When to Use This Tool +Use this tool proactively in these scenarios: + +1. Complex multi-step tasks - When a task requires 3 or more distinct steps or actions +2. Non-trivial and complex tasks - Tasks that require careful planning or multiple operations +3. User explicitly requests todo list - When the user directly asks you to use the todo list +4. User provides multiple tasks - When users provide a list of things to be done (numbered or comma-separated) +5. After receiving new instructions - Immediately capture user requirements as todos +6. When you start working on a task - Mark it as in_progress BEFORE beginning work. Ideally you should only have one todo as in_progress at a time +7. After completing a task - Mark it as completed and add any new follow-up tasks discovered during implementation + +## When NOT to Use This Tool + +Skip using this tool when: +1. There is only a single, straightforward task +2. The task is trivial and tracking it provides no organizational benefit +3. The task can be completed in less than 3 trivial steps +4. The task is purely conversational or informational + +NOTE that you should not use this tool if there is only one trivial task to do. In this case you are better off just doing the task directly. + +## Examples of When to Use the Todo List + + +User: I want to add a dark mode toggle to the application settings. Make sure you run the tests and build when you're done! +Assistant: I'll help add a dark mode toggle to your application settings. Let me create a todo list to track this implementation. +*Creates todo list with the following items:* +1. Create dark mode toggle component in Settings page +2. Add dark mode state management (context/store) +3. Implement CSS-in-JS styles for dark theme +4. Update existing components to support theme switching +5. Run tests and build process, addressing any failures or errors that occur +*Begins working on the first task* + + +The assistant used the todo list because: +1. Adding dark mode is a multi-step feature requiring UI, state management, and styling changes +2. The user explicitly requested tests and build be run afterward +3. The assistant inferred that tests and build need to pass by adding "Ensure tests and build succeed" as the final task + + + + +User: Help me rename the function getCwd to getCurrentWorkingDirectory across my project +Assistant: Let me first search through your codebase to find all occurrences of 'getCwd'. +*Uses grep or search tools to locate all instances of getCwd in the codebase* +Assistant: I've found 15 instances of 'getCwd' across 8 different files. Let me create a todo list to track these changes. +*Creates todo list with specific items for each file that needs updating* + + +The assistant used the todo list because: +1. First, the assistant searched to understand the scope of the task +2. Upon finding multiple occurrences across different files, it determined this was a complex task with multiple steps +3. The todo list helps ensure every instance is tracked and updated systematically +4. This approach prevents missing any occurrences and maintains code consistency + + + + + +User: I need to implement these features for my e-commerce site: user registration, product catalog, shopping cart, and checkout flow. +Assistant: I'll help implement these features. First, let's add all the features to the todo list. +*Creates a todo list breaking down each feature into specific tasks based on the project architecture* +Assistant: Let's start with user registration. This will involve creating database models, API endpoints, and frontend forms that integrate with your existing authentication system. + + +The assistant used the todo list because: +1. The user provided multiple complex features to implement in a comma separated list +2. The todo list helps organize these large features into manageable tasks +3. This approach allows for tracking progress across the entire implementation + + + + +User: Can you help optimize my React application? It's rendering slowly and has performance issues. +Assistant: I'll help optimize your React application. First, let me examine your codebase to identify potential performance bottlenecks. +*Reviews component structure, render patterns, state management, and data fetching* +Assistant: After analyzing your codebase, I've identified several performance issues. Let me create a todo list to track our optimization efforts. +*Creates todo list with items like: 1) Implement memoization for expensive calculations in ProductList, 2) Add virtualization for long lists in Dashboard, 3) Optimize image loading in Gallery component, 4) Fix state update loops in ShoppingCart, 5) Review bundle size and implement code splitting* +Let's start by implementing memoization for the expensive calculations in your ProductList component. + + +The assistant used the todo list because: +1. First, the assistant examined the codebase to identify specific performance issues +2. Based on this analysis, it identified multiple optimization opportunities +3. Performance optimization is a non-trivial task requiring multiple steps +4. The todo list helps methodically track improvements across different components +5. This systematic approach ensures all performance bottlenecks are addressed + + + +## Examples of When NOT to Use the Todo List + + +User: How do I print 'Hello World' in Python? +Assistant: In Python, you can print "Hello World" with this simple code: + +python +print("Hello World") + +This will output the text "Hello World" to the console when executed. + + +The assistant did not use the todo list because this is a single, trivial task that can be completed in one step. There's no need to track multiple tasks or steps for such a straightforward request. + + + + +User: What does the git status command do? +Assistant: The git status command shows the current state of your working directory and staging area. It displays which changes have been staged, which haven't, and which files aren't being tracked by Git. This is a fundamental command for understanding the current state of your repository before making commits. + + +The assistant did not use the todo list because this is an informational request with no actual coding task to complete. The user is simply asking for an explanation, not for the assistant to perform multiple steps or tasks. + + + + +User: Can you add a comment to the calculateTotal function to explain what it does? +Assistant: Sure, let me add a comment to the calculateTotal function to explain what it does. +* Uses the Edit tool to add a comment to the calculateTotal function * + + +The assistant did not use the todo list because this is a single, straightforward task confined to one location in the code. Adding a comment doesn't require tracking multiple steps or systematic organization. + + + + +User: Run npm install for me and tell me what happens. +Assistant: I'll run the npm install command for you. + +*Executes: npm install* + +The command completed successfully. Here's the output: +[Output of npm install command] + +All dependencies have been installed according to your package.json file. + + +The assistant did not use the todo list because this is a single command execution with immediate results. There are no multiple steps to track or organize, making the todo list unnecessary for this straightforward task. + + + +## Task States and Management + +1. **Task States**: Use these states to track progress: + - pending: Task not yet started + - in_progress: Currently working on (limit to ONE task at a time) + - completed: Task finished successfully + +2. **Task Management**: + - Update task status in real-time as you work + - Mark tasks complete IMMEDIATELY after finishing (don't batch completions) + - Only have ONE task in_progress at any time + - Complete current tasks before starting new ones + - Remove tasks that are no longer relevant from the list entirely + +3. **Task Completion Requirements**: + - ONLY mark a task as completed when you have FULLY accomplished it + - If you encounter errors, blockers, or cannot finish, keep the task as in_progress + - When blocked, create a new task describing what needs to be resolved + - Never mark a task as completed if: + - Tests are failing + - Implementation is partial + - You encountered unresolved errors + - You couldn't find necessary files or dependencies + +4. **Task Breakdown**: + - Create specific, actionable items + - Break complex tasks into smaller, manageable steps + - Use clear, descriptive task names + +When in doubt, use this tool. Being proactive with task management demonstrates attentiveness and ensures you complete all requirements successfully. + + +```typescript +{ + // The updated todo list + todos: { + content: string; + status: "pending" | "in_progress" | "completed"; + priority: "high" | "medium" | "low"; + id: string; + }[]; +} +``` diff --git a/tests/core/test_todo_manager.py b/tests/core/test_todo_manager.py new file mode 100644 index 0000000000..a05f5aa1d0 --- /dev/null +++ b/tests/core/test_todo_manager.py @@ -0,0 +1,108 @@ +from holmes.core.todo_manager import ( + TodoListManager, + get_todo_manager, + set_current_session_id, + get_session_id_from_context, +) +from holmes.core.tools import Task, TaskStatus + + +class TestTodoListManager: + def test_manager_creation(self): + """Test that TodoListManager can be created.""" + manager = TodoListManager() + assert manager.get_session_count() == 0 + + def test_empty_session(self): + """Test handling of empty session.""" + manager = TodoListManager() + tasks = manager.get_session_tasks("non-existent-session") + assert tasks == [] + + prompt_context = manager.format_tasks_for_prompt("non-existent-session") + assert prompt_context == "" + + def test_session_task_management(self): + """Test adding and retrieving tasks from session.""" + manager = TodoListManager() + session_id = "test-session" + + # Create test tasks + tasks = [ + Task(id="1", content="Task 1", status=TaskStatus.PENDING), + Task(id="2", content="Task 2", status=TaskStatus.IN_PROGRESS), + Task(id="3", content="Task 3", status=TaskStatus.COMPLETED), + ] + + # Update session tasks + manager.update_session_tasks(session_id, tasks) + assert manager.get_session_count() == 1 + + # Retrieve tasks + retrieved_tasks = manager.get_session_tasks(session_id) + assert len(retrieved_tasks) == 3 + assert retrieved_tasks[0].content == "Task 1" + assert retrieved_tasks[1].content == "Task 2" + assert retrieved_tasks[2].content == "Task 3" + + def test_prompt_formatting(self): + """Test that tasks are formatted correctly for prompt injection.""" + manager = TodoListManager() + session_id = "test-prompt-session" + + tasks = [ + Task(id="1", content="Check system", status=TaskStatus.PENDING), + Task(id="2", content="Review logs", status=TaskStatus.IN_PROGRESS), + Task(id="3", content="Write report", status=TaskStatus.COMPLETED), + ] + + manager.update_session_tasks(session_id, tasks) + prompt_context = manager.format_tasks_for_prompt(session_id) + + # Check structure + assert "# CURRENT INVESTIGATION TASKS" in prompt_context + assert ( + "**Task Status**: 1 completed, 1 in progress, 1 pending" in prompt_context + ) + + # Check tasks appear + assert "Check system" in prompt_context + assert "Review logs" in prompt_context + assert "Write report" in prompt_context + + # Check indicators + assert "[ ]" in prompt_context # pending + assert "[~]" in prompt_context # in_progress + assert "[✓]" in prompt_context # completed + + # Check priority + assert "(HIGH)" in prompt_context + assert "(MED)" in prompt_context + assert "(LOW)" in prompt_context + + def test_session_clearing(self): + """Test clearing session tasks.""" + manager = TodoListManager() + session_id = "test-clear-session" + + tasks = [Task(id="1", content="Task 1", status=TaskStatus.PENDING)] + manager.update_session_tasks(session_id, tasks) + assert manager.get_session_count() == 1 + + manager.clear_session(session_id) + assert manager.get_session_count() == 0 + assert manager.get_session_tasks(session_id) == [] + + def test_global_manager_instance(self): + """Test that get_todo_manager returns the same instance.""" + manager1 = get_todo_manager() + manager2 = get_todo_manager() + assert manager1 is manager2 + + def test_session_id_management(self): + """Test session ID context management.""" + test_session_id = "context-test-session" + set_current_session_id(test_session_id) + + retrieved_session_id = get_session_id_from_context() + assert retrieved_session_id == test_session_id diff --git a/tests/core/test_todo_write_tool.py b/tests/core/test_todo_write_tool.py new file mode 100644 index 0000000000..8e3b8c75cf --- /dev/null +++ b/tests/core/test_todo_write_tool.py @@ -0,0 +1,183 @@ +from holmes.core.tools import TodoWriteTool, ToolResultStatus, TaskStatus, TaskPriority + + +class TestTodoWriteTool: + def test_todo_write_tool_creation(self): + """Test that TodoWriteTool can be created with correct parameters.""" + tool = TodoWriteTool() + assert tool.name == "TodoWrite" + assert "investigation tasks" in tool.description + assert "todos" in tool.parameters + + def test_todo_write_tool_empty_params(self): + """Test TodoWriteTool with empty parameters.""" + tool = TodoWriteTool() + result = tool._invoke({}) + + assert result.status == ToolResultStatus.SUCCESS + assert isinstance(result.data, str) + assert "0 tasks" in result.data + assert "Investigation plan updated" in result.data + + def test_todo_write_tool_with_tasks(self): + """Test TodoWriteTool with valid task data.""" + tool = TodoWriteTool() + params = { + "todos": [ + { + "id": "1", + "content": "Check pod status", + "status": "pending", + "priority": "high", + }, + { + "id": "2", + "content": "Analyze logs", + "status": "in_progress", + "priority": "medium", + }, + ] + } + + result = tool._invoke(params) + + assert result.status == ToolResultStatus.SUCCESS + assert isinstance(result.data, str) + assert "2 tasks" in result.data + assert "Investigation plan updated" in result.data + # Should include pretty printed TodoList + assert "Check pod status" in result.data + assert "Analyze logs" in result.data + + def test_todo_write_tool_default_values(self): + """Test TodoWriteTool with minimal task data uses defaults.""" + tool = TodoWriteTool() + params = {"todos": [{"content": "Test task"}]} + + result = tool._invoke(params) + + assert result.status == ToolResultStatus.SUCCESS + assert isinstance(result.data, str) + assert "1 tasks" in result.data + assert "Investigation plan updated" in result.data + # Should include pretty printed TodoList + assert "Test task" in result.data + + def test_todo_write_tool_invalid_enum_values(self): + """Test TodoWriteTool handles invalid enum values gracefully.""" + tool = TodoWriteTool() + params = { + "todos": [ + { + "content": "Test task", + "status": "invalid_status", + "priority": "invalid_priority", + } + ] + } + + result = tool._invoke(params) + + # Should handle gracefully and return error + assert result.status == ToolResultStatus.ERROR + assert "Failed to process tasks" in result.error + + def test_get_parameterized_one_liner(self): + """Test the parameterized one-liner description.""" + tool = TodoWriteTool() + + params = {"todos": [{"content": "task1"}, {"content": "task2"}]} + one_liner = tool.get_parameterized_one_liner(params) + assert "2 investigation tasks" in one_liner + + params = {"todos": []} + one_liner = tool.get_parameterized_one_liner(params) + assert "0 investigation tasks" in one_liner + + def test_task_status_enum(self): + """Test TaskStatus enum values.""" + assert TaskStatus.PENDING == "pending" + assert TaskStatus.IN_PROGRESS == "in_progress" + assert TaskStatus.COMPLETED == "completed" + + def test_task_priority_enum(self): + """Test TaskPriority enum values.""" + assert TaskPriority.HIGH == "high" + assert TaskPriority.MEDIUM == "medium" + assert TaskPriority.LOW == "low" + + def test_openai_format(self): + """Test that the tool generates correct OpenAI format.""" + tool = TodoWriteTool() + openai_format = tool.get_openai_format() + + assert openai_format["type"] == "function" + assert openai_format["function"]["name"] == "TodoWrite" + assert "investigation tasks" in openai_format["function"]["description"] + + # Check parameters schema + params = openai_format["function"]["parameters"] + assert params["type"] == "object" + assert "todos" in params["properties"] + + # Check array schema has items property + todos_param = params["properties"]["todos"] + assert todos_param["type"] == "array" + assert "items" in todos_param + assert todos_param["items"]["type"] == "object" + + # Check required fields + assert "todos" in params["required"] + + def test_session_storage_functionality(self): + """Test that the tool stores tasks in session storage.""" + from holmes.core.todo_manager import get_todo_manager, set_current_session_id + + tool = TodoWriteTool() + session_id = "test-session-456" + set_current_session_id(session_id) + + params = { + "todos": [ + { + "id": "1", + "content": "Task 1", + "status": "pending", + "priority": "high", + }, + { + "id": "2", + "content": "Task 2", + "status": "completed", + "priority": "low", + }, + ] + } + + result = tool._invoke(params) + + assert result.status == ToolResultStatus.SUCCESS + assert "2 tasks" in result.data + assert "Investigation plan updated" in result.data + + # Check that the pretty printed TodoList is included in the response + assert "CURRENT INVESTIGATION TASKS" in result.data + assert "Task 1" in result.data + assert "Task 2" in result.data + assert "[ ]" in result.data # pending indicator + assert "[✓]" in result.data # completed indicator + assert "(HIGH)" in result.data # priority indicator + assert "(LOW)" in result.data # priority indicator + + # Check session storage + manager = get_todo_manager() + stored_tasks = manager.get_session_tasks(session_id) + assert len(stored_tasks) == 2 + assert stored_tasks[0].content == "Task 1" + assert stored_tasks[1].content == "Task 2" + + # Check prompt formatting matches what's in the response + prompt_context = manager.format_tasks_for_prompt(session_id) + assert ( + prompt_context in result.data + ) # The formatted tasks should be part of the response diff --git a/tests/plugins/toolsets/test_core_investigation.py b/tests/plugins/toolsets/test_core_investigation.py new file mode 100644 index 0000000000..361c96a616 --- /dev/null +++ b/tests/plugins/toolsets/test_core_investigation.py @@ -0,0 +1,41 @@ +from holmes.plugins.toolsets.investigator.core_investigation import ( + CoreInvestigationToolset, +) +from holmes.core.tools import ToolsetStatusEnum, ToolsetTag, TodoWriteTool + + +class TestCoreInvestigationToolset: + def test_toolset_creation(self): + """Test that CoreInvestigationToolset is created correctly.""" + toolset = CoreInvestigationToolset() + + assert toolset.name == "core_investigation" + assert "investigation tools" in toolset.description + assert toolset.enabled is True + assert toolset.is_default is True + assert ToolsetTag.CORE in toolset.tags + + def test_toolset_has_todo_write_tool(self): + """Test that the toolset includes the TodoWrite tool.""" + toolset = CoreInvestigationToolset() + + assert len(toolset.tools) == 1 + assert isinstance(toolset.tools[0], TodoWriteTool) + assert toolset.tools[0].name == "TodoWrite" + + def test_toolset_check_prerequisites(self): + """Test that toolset prerequisites check passes.""" + toolset = CoreInvestigationToolset() + toolset.check_prerequisites() + + # Should be enabled by default with no prerequisites + assert toolset.status == ToolsetStatusEnum.ENABLED + assert toolset.error is None + + def test_get_example_config(self): + """Test that example config is returned.""" + toolset = CoreInvestigationToolset() + config = toolset.get_example_config() + + assert isinstance(config, dict) + # Core toolset doesn't need configuration From 3b9f95f708ed8ce311bace8edc1bdf21a7835086 Mon Sep 17 00:00:00 2001 From: Arik Alon Date: Tue, 12 Aug 2025 15:12:23 +0300 Subject: [PATCH 04/17] improvements --- holmes/core/todo_manager.py | 3 +- holmes/core/tools.py | 23 ++++++------- holmes/plugins/prompts/generic_ask.jinja2 | 33 ++++++++++++++++++- .../test_case.yaml | 9 +++++ .../toolsets.yaml | 3 ++ 5 files changed, 56 insertions(+), 15 deletions(-) create mode 100644 tests/llm/fixtures/test_ask_holmes/150_what_is_wrong_with_mimir/test_case.yaml create mode 100644 tests/llm/fixtures/test_ask_holmes/150_what_is_wrong_with_mimir/toolsets.yaml diff --git a/holmes/core/todo_manager.py b/holmes/core/todo_manager.py index b80722406c..ba661fc0a1 100644 --- a/holmes/core/todo_manager.py +++ b/holmes/core/todo_manager.py @@ -108,7 +108,8 @@ def get_session_id_from_context(context=None) -> str: # Generate new session ID session_id = str(uuid4()) - context._current_session = session_id + if context: + context._current_session = session_id return session_id diff --git a/holmes/core/tools.py b/holmes/core/tools.py index 344cd47e2f..f99b86ca3b 100644 --- a/holmes/core/tools.py +++ b/holmes/core/tools.py @@ -602,12 +602,13 @@ class TodoWriteTool(Tool): } def _invoke(self, params: Dict) -> StructuredToolResult: - try: - from holmes.core.todo_manager import ( - get_todo_manager, - get_session_id_from_context, - ) + from holmes.core.todo_manager import ( + get_todo_manager, + get_session_id_from_context, + ) + try: + logging.info(f"### invoking todowrite tool {params}") todos_data = params.get("todos", []) tasks = [] @@ -638,15 +639,11 @@ def print_tasks_table(tasks): max_id_width = max(len(str(task.id)) for task in tasks) max_content_width = max(len(task.content) for task in tasks) max_status_width = max(len(task.status.value) for task in tasks) - max_priority_width = max( - len(task.priority.value.upper()) for task in tasks - ) # Ensure minimum widths for headers id_width = max(max_id_width, 2) content_width = max(max_content_width, 7) status_width = max(max_status_width, 6) - priority_width = max(max_priority_width, 8) # Status indicators status_icons = { @@ -656,8 +653,8 @@ def print_tasks_table(tasks): } # Build table - separator = f"+{'-' * (id_width + 2)}+{'-' * (content_width + 2)}+{'-' * (status_width + 2)}+{'-' * (priority_width + 2)}+" - header = f"| {'ID':<{id_width}} | {'Content':<{content_width}} | {'Status':<{status_width}} | {'Priority':<{priority_width}} |" + separator = f"+{'-' * (id_width + 2)}+{'-' * (content_width + 2)}+{'-' * (status_width + 2)}+" + header = f"| {'ID':<{id_width}} | {'Content':<{content_width}} | {'Status':<{status_width}} |" # Log the table logging.info("Updated Investigation Tasks:") @@ -669,8 +666,7 @@ def print_tasks_table(tasks): status_display = ( f"{status_icons[task.status.value]} {task.status.value}" ) - priority_display = task.priority.value.upper() - row = f"| {task.id:<{id_width}} | {task.content:<{content_width}} | {status_display:<{status_width}} | {priority_display:<{priority_width}} |" + row = f"| {task.id:<{id_width}} | {task.content:<{content_width}} | {status_display:<{status_width}} |" logging.info(row) logging.info(separator) @@ -695,6 +691,7 @@ def print_tasks_table(tasks): ) except Exception as e: + logging.exception("error using todowrite tool") return StructuredToolResult( status=ToolResultStatus.ERROR, error=f"Failed to process tasks: {str(e)}", diff --git a/holmes/plugins/prompts/generic_ask.jinja2 b/holmes/plugins/prompts/generic_ask.jinja2 index 9103635824..301aca1024 100644 --- a/holmes/plugins/prompts/generic_ask.jinja2 +++ b/holmes/plugins/prompts/generic_ask.jinja2 @@ -5,7 +5,38 @@ Do not say 'based on the tool output' or explicitly refer to tools at all. If you output an answer and then realize you need to call more tools or there are possible next steps, you may do so by calling tools at that point in time. If you have a good and concrete suggestion for how the user can fix something, tell them even if not asked explicitly -CRITICAL: For multi-step questions, you MUST start by calling the TodoWrite tool to create your investigation plan. +CRITICAL: For multi-step questions, you MUST start by calling the TodoWrite tool with a `todos` parameter containing an array of task objects. Each task must have: +- `id`: unique identifier (string) +- `content`: specific task description (string) +- `status`: "pending" for new tasks (string) + +MANDATORY Task Status Updates: +- When starting a task: Call TodoWrite changing that task's status to "in_progress" +- When completing a task: Call TodoWrite changing that task's status to "completed" +- Keep all other tasks unchanged in each TodoWrite call +- Only work on ONE task at a time (only one "in_progress" task) + +# MANDATORY Self-Verification Checklist + +Before providing any final answer, you MUST create and complete a verification task using TodoWrite that checks: +✓ **Completeness**: Does my answer fully address what the user asked? +✓ **Specificity**: Have I provided exact resource names, namespaces, versions, and commands? +✓ **Root Cause**: Did I use "five whys" to find the actual root cause, not just symptoms? +✓ **Actionability**: Have I given concrete next steps the user can take? +✓ **Evidence**: Is every claim in my answer backed by tool output or investigation data? +✓ **Assumptions**: Have I verified all resource names, namespaces, and configurations through tools rather than assuming? + +Only after completing this verification checklist should you provide your final answer. + + Status Update Example: + # Starting task 2: + TodoWrite(todos=[ + {"id": "1", "content": "Check pod status", "status": "completed"}, + {"id": "2", "content": "Examine logs", "status": "in_progress"}, + {"id": "3", "content": "Check resources", "status": "pending"} + {"id": "4", "content": "Investigation Verification", "status": "pending"} + ]) + {% if todo_list %} {{ todo_list }} diff --git a/tests/llm/fixtures/test_ask_holmes/150_what_is_wrong_with_mimir/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/150_what_is_wrong_with_mimir/test_case.yaml new file mode 100644 index 0000000000..621ab86687 --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/150_what_is_wrong_with_mimir/test_case.yaml @@ -0,0 +1,9 @@ +user_prompt: + - "What is wrong with mimir?" +expected_output: | + Output must contain 1 or more embeds in the following format + <<{"type": "datadogql", "tool_name": "query_datadog_metrics", "random_key": "iD8G"}>> + + random_key may be different than the above example, but all other parameters (type and tool_name) must be as described + + Output must NOT tell the user it doesn't have access to metrics or that they should use another tool diff --git a/tests/llm/fixtures/test_ask_holmes/150_what_is_wrong_with_mimir/toolsets.yaml b/tests/llm/fixtures/test_ask_holmes/150_what_is_wrong_with_mimir/toolsets.yaml new file mode 100644 index 0000000000..3c9a2dc247 --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/150_what_is_wrong_with_mimir/toolsets.yaml @@ -0,0 +1,3 @@ +toolsets: + kubernetes/core: + enabled: true From 5f31e18399a1d64bab0cbd2036f95a05f598e501 Mon Sep 17 00:00:00 2001 From: Arik Alon Date: Tue, 12 Aug 2025 19:08:15 +0300 Subject: [PATCH 05/17] prompt changes --- holmes/core/tool_calling_llm.py | 2 +- holmes/plugins/prompts/generic_ask.jinja2 | 54 +---- .../prompts/investigation_procedure.jinja2 | 206 ++++++++++++++++++ 3 files changed, 208 insertions(+), 54 deletions(-) create mode 100644 holmes/plugins/prompts/investigation_procedure.jinja2 diff --git a/holmes/core/tool_calling_llm.py b/holmes/core/tool_calling_llm.py index 9726491cf6..3944923fef 100644 --- a/holmes/core/tool_calling_llm.py +++ b/holmes/core/tool_calling_llm.py @@ -204,7 +204,7 @@ def __init__( self, tool_executor: ToolExecutor, max_steps: int, llm: LLM, tracer=None ): self.tool_executor = tool_executor - self.max_steps = max_steps + self.max_steps = 40 ## TODO Arik - remove this self.tracer = tracer self.llm = llm diff --git a/holmes/plugins/prompts/generic_ask.jinja2 b/holmes/plugins/prompts/generic_ask.jinja2 index 301aca1024..bf5af2e130 100644 --- a/holmes/plugins/prompts/generic_ask.jinja2 +++ b/holmes/plugins/prompts/generic_ask.jinja2 @@ -5,60 +5,8 @@ Do not say 'based on the tool output' or explicitly refer to tools at all. If you output an answer and then realize you need to call more tools or there are possible next steps, you may do so by calling tools at that point in time. If you have a good and concrete suggestion for how the user can fix something, tell them even if not asked explicitly -CRITICAL: For multi-step questions, you MUST start by calling the TodoWrite tool with a `todos` parameter containing an array of task objects. Each task must have: -- `id`: unique identifier (string) -- `content`: specific task description (string) -- `status`: "pending" for new tasks (string) +{% include 'investigation_procedure.jinja2' %} -MANDATORY Task Status Updates: -- When starting a task: Call TodoWrite changing that task's status to "in_progress" -- When completing a task: Call TodoWrite changing that task's status to "completed" -- Keep all other tasks unchanged in each TodoWrite call -- Only work on ONE task at a time (only one "in_progress" task) - -# MANDATORY Self-Verification Checklist - -Before providing any final answer, you MUST create and complete a verification task using TodoWrite that checks: -✓ **Completeness**: Does my answer fully address what the user asked? -✓ **Specificity**: Have I provided exact resource names, namespaces, versions, and commands? -✓ **Root Cause**: Did I use "five whys" to find the actual root cause, not just symptoms? -✓ **Actionability**: Have I given concrete next steps the user can take? -✓ **Evidence**: Is every claim in my answer backed by tool output or investigation data? -✓ **Assumptions**: Have I verified all resource names, namespaces, and configurations through tools rather than assuming? - -Only after completing this verification checklist should you provide your final answer. - - Status Update Example: - # Starting task 2: - TodoWrite(todos=[ - {"id": "1", "content": "Check pod status", "status": "completed"}, - {"id": "2", "content": "Examine logs", "status": "in_progress"}, - {"id": "3", "content": "Check resources", "status": "pending"} - {"id": "4", "content": "Investigation Verification", "status": "pending"} - ]) - - -{% if todo_list %} -{{ todo_list }} -{% endif %} - -# MANDATORY Investigation Process - -For ANY question that requires multiple steps or investigation, you MUST follow this structured approach: - -1. **IMMEDIATELY START with TodoWrite**: Before doing any investigation, you MUST call the TodoWrite tool to create an investigation plan. Break down the question into specific tasks. - -2. **Problem Analysis**: Identify what sub-problems need to be solved and create tasks for each. - -3. **Task Execution**: Execute your plan systematically. You MUST: - - Mark tasks as 'in_progress' when you start working on them - - Call TodoWrite to update task status as you complete each task - - Reference the visible task list from your TodoWrite output throughout your investigation - - Follow the exact tasks you created - complete all of them - -4. **Verification**: Before providing your final answer, verify that your conclusion is complete and addresses the user's question fully. - -IMPORTANT: If a question requires more than one investigation step, your FIRST tool call MUST be TodoWrite. {% include '_current_date_time.jinja2' %} Use conversation history to maintain continuity when appropriate, ensuring efficiency in your responses. diff --git a/holmes/plugins/prompts/investigation_procedure.jinja2 b/holmes/plugins/prompts/investigation_procedure.jinja2 new file mode 100644 index 0000000000..34cab4a285 --- /dev/null +++ b/holmes/plugins/prompts/investigation_procedure.jinja2 @@ -0,0 +1,206 @@ +CRITICAL: For multi-step questions, you MUST start by calling the TodoWrite tool with a `todos` parameter containing an array of task objects. Each task must have: +- `id`: unique identifier (string) +- `content`: specific task description (string) +- `status`: "pending" for new tasks (string) + +MANDATORY Task Status Updates: +- When starting a task: Call TodoWrite changing that task's status to "in_progress" +- When completing a task: Call TodoWrite changing that task's status to "completed" +- Keep all other tasks unchanged in each TodoWrite call +- Only work on ONE task at a time (only one "in_progress" task) + +# MANDATORY Self-Verification Checklist + +Before providing any final answer, you MUST create and complete a verification task using TodoWrite that checks: +✓ **Completeness**: Does my answer fully address what the user asked? +✓ **Specificity**: Have I provided exact resource names, namespaces, versions, and commands? +✓ **Root Cause**: Did I use "five whys" to find the actual root cause, not just symptoms? +✓ **Actionability**: Have I given concrete next steps the user can take? +✓ **Evidence**: Is every claim in my answer backed by tool output or investigation data? +✓ **Assumptions**: Have I verified all resource names, namespaces, and configurations through tools rather than assuming? + +Only after completing this verification checklist should you provide your final answer. + +# CRITICAL: TASK COMPLETION ENFORCEMENT + +YOU MUST COMPLETE EVERY SINGLE TASK before providing your final answer. NO EXCEPTIONS. + +**BEFORE providing any final answer or conclusion, you MUST:** + +1. **Check TodoWrite status**: Verify ALL tasks show "completed" status +2. **If ANY task is "pending" or "in_progress"**: + - DO NOT provide a final answer + - Continue working on the next pending task + - Use TodoWrite to mark it "in_progress" + - Complete the task + - Mark it "completed" with TodoWrite +3. **Only after ALL tasks are "completed"**: Proceed to verification and final answer + +**VIOLATION CONSEQUENCES**: +- Providing answers with pending tasks = INVESTIGATION FAILURE +- You MUST complete the verification task as the final step before any answer +- Incomplete investigations are unacceptable and must be continued + +**Task Status Check Example:** +Before final answer, confirm you see something like: +[✓] completed - Task 1 +[✓] completed - Task 2[✓] completed - Task 3 +[✓] completed - Investigation Verification + +If you see ANY `[ ] pending` or `[~] in_progress` tasks, DO NOT provide final answer. + + Status Update Example: + # Starting task 2: + TodoWrite(todos=[ + {"id": "1", "content": "Check pod status", "status": "completed"}, + {"id": "2", "content": "Examine logs", "status": "in_progress"}, + {"id": "3", "content": "Check resources", "status": "pending"} + ]) + + +{% if todo_list %} +{{ todo_list }} +{% endif %} + +# MANDATORY Investigation Process + +For ANY question that requires multiple steps or investigation, you MUST follow this structured approach: + +1. **IMMEDIATELY START with TodoWrite**: Before doing any investigation, you MUST call the TodoWrite tool to create an investigation plan. Break down the question into specific tasks. + +2. **Problem Analysis**: Identify what sub-problems need to be solved and create tasks for each. + +3. **Task Execution**: Execute your plan systematically. You MUST: + - Mark tasks as 'in_progress' when you start working on them + - Call TodoWrite to update task status as you complete each task + - Reference the visible task list from your TodoWrite output throughout your investigation + - Follow the exact tasks you created - complete all of them + +4. **Verification**: Before providing your final answer, verify that your conclusion is complete and addresses the user's question fully. + +IMPORTANT: If a question requires more than one investigation step, your FIRST tool call MUST be TodoWrite. + +# MANDATORY Multi-Phase Investigation Process + +For ANY question requiring investigation, you MUST follow this structured approach: + +## Phase 1: Initial Investigation +1. **IMMEDIATELY START with TodoWrite**: Create initial investigation task list +2. **Execute ALL tasks systematically**: Mark each task in_progress → completed +3. **Complete EVERY task** in the current list before proceeding + +## Phase Evaluation and Continuation +After completing ALL tasks in current list, you MUST: + +1. **STOP and Evaluate**: Ask yourself these critical questions: + - "Do I have enough information to completely answer the user's question?" + - "Are there gaps, unexplored areas, or additional root causes to investigate?" + - "Have I followed the 'five whys' methodology to the actual root cause?" + - "Did my investigation reveal new questions or areas that need exploration?" + +2. **If Investigation is INCOMPLETE**: + - Call TodoWrite to create a NEW task list for the next investigation phase + - Label it clearly: "Investigation Phase 2: [specific focus area]" + - Focus tasks on the specific gaps/questions discovered in the previous phase + - Execute ALL tasks in this new list + - Repeat this evaluation process + +3. **Continue Creating New Phases** until you can answer "YES" to: + - "I have thoroughly investigated all aspects of this problem" + - "I can provide a complete answer with specific, actionable information" + - "No additional investigation would significantly improve my answer" + +## MANDATORY Final Phase: Verification +When investigation is complete, you MUST create one final verification phase: + +1. **Create Verification Task List** using TodoWrite with these EXACT tasks: + TodoWrite(todos=[ + {"id": "v1", "content": "Verify my answer completely addresses the original user question", "status": "pending"}, + {"id": "v2", "content": "Confirm all claims in my answer are backed by tool output evidence", "status": "pending"}, + {"id": "v3", "content": "Validate root cause analysis follows five whys methodology correctly", "status": "pending"}, + {"id": "v4", "content": "Check that specific actionable information is provided (exact names, commands, steps)", "status": "pending"}, + {"id": "v5", "content": "Ensure no assumptions were made without verification through tools", "status": "pending"}, + {"id": "v6", "content": "Self-critique: identify any weaknesses or gaps in my proposed answer", "status": "pending"} + ]) + +2. **Execute ALL verification tasks** systematically before final answer + +## CRITICAL ENFORCEMENT RULES + + **ABSOLUTE REQUIREMENTS:** + - NO final answer until verification phase is 100% completed + - Each investigation phase must have ALL tasks completed before evaluation + - You MUST explicitly create new investigation phases when gaps are identified + - Verification phase is MANDATORY - never skip it + - Mark each task as "in_progress" when working on it, "completed" when done + + **EXAMPLES of Phase Progression:** + + *Phase 1*: Initial investigation discovers pod crashes + *Phase 2*: Deep dive into specific pod logs and resource constraints + *Phase 3*: Investigate upstream services causing the crashes + *Verification Phase*: Self-critique and validate the complete solution + + **VIOLATION CONSEQUENCES:** + - Providing answers without verification phase = INVESTIGATION FAILURE + - Skipping investigation phases when gaps exist = INCOMPLETE ANALYSIS + - Not completing all tasks in a phase = PROCESS VIOLATION + +# VERIFICATION PHASE EXECUTION GUIDE + + When executing verification tasks, you must: + + **For Task v1 (Answer Completeness):** + - Reread the original user question word-by-word + - Compare against your proposed answer + - Identify any aspects not addressed + + **For Task v2 (Evidence Backing):** + - List each claim in your answer + - Trace each claim back to specific tool outputs + - Flag any unsupported statements + + **For Task v3 (Root Cause Analysis):** + - Walk through your "five whys" chain + - Verify each "why" logically follows from evidence + - Ensure you reached actual root cause, not just symptoms + + **For Task v4 (Actionable Information):** + - Verify exact resource names are provided (not generic examples) + - Check commands are complete and runnable + - Ensure steps are specific to user's environment + + **For Task v5 (No Assumptions):** + - List any resource names, namespaces, configurations mentioned + - Verify each was confirmed via tool calls + - Flag anything assumed without verification + + **For Task v6 (Self-Critique):** + - Identify potential weaknesses in your investigation + - Consider alternative explanations not explored + - Assess if additional investigation would strengthen answer + + +# INVESTIGATION PHASE TRANSITION EXAMPLES + + **Example 1: Network Connectivity Issue** + Phase 1: Check pod status, basic connectivity + → Evaluation: Found network issues, but need to investigate underlying cause + Phase 2: Investigate network policies, DNS resolution + → Evaluation: Found DNS issues, need to check DNS server configuration + Phase 3: Deep dive into DNS server pods, configuration + → Evaluation: Complete - found misconfigured DNS upstream + Verification Phase: Validate solution addresses original connectivity problem + + **Example 2: Application Performance Issue** + Phase 1: Check application metrics, resource usage + → Evaluation: Found high CPU usage, but root cause unclear + Phase 2: Investigate database connections, query performance + → Evaluation: Complete - found slow database queries causing CPU spike + Verification Phase: Confirm analysis provides actionable database optimization steps + + **REMEMBER:** Each evaluation is a decision point: + - Continue investigating (create new phase) OR + - Proceed to verification (investigation complete) + + Never guess - if unsure whether investigation is complete, create another phase. From cde09328f19c5f34072507a6cb2a1a781e79c9d8 Mon Sep 17 00:00:00 2001 From: Arik Alon Date: Tue, 12 Aug 2025 20:36:27 +0300 Subject: [PATCH 06/17] prompt changes --- .../prompts/investigation_procedure.jinja2 | 48 ++----------------- 1 file changed, 5 insertions(+), 43 deletions(-) diff --git a/holmes/plugins/prompts/investigation_procedure.jinja2 b/holmes/plugins/prompts/investigation_procedure.jinja2 index 34cab4a285..e369c860bb 100644 --- a/holmes/plugins/prompts/investigation_procedure.jinja2 +++ b/holmes/plugins/prompts/investigation_procedure.jinja2 @@ -9,18 +9,6 @@ MANDATORY Task Status Updates: - Keep all other tasks unchanged in each TodoWrite call - Only work on ONE task at a time (only one "in_progress" task) -# MANDATORY Self-Verification Checklist - -Before providing any final answer, you MUST create and complete a verification task using TodoWrite that checks: -✓ **Completeness**: Does my answer fully address what the user asked? -✓ **Specificity**: Have I provided exact resource names, namespaces, versions, and commands? -✓ **Root Cause**: Did I use "five whys" to find the actual root cause, not just symptoms? -✓ **Actionability**: Have I given concrete next steps the user can take? -✓ **Evidence**: Is every claim in my answer backed by tool output or investigation data? -✓ **Assumptions**: Have I verified all resource names, namespaces, and configurations through tools rather than assuming? - -Only after completing this verification checklist should you provide your final answer. - # CRITICAL: TASK COMPLETION ENFORCEMENT YOU MUST COMPLETE EVERY SINGLE TASK before providing your final answer. NO EXCEPTIONS. @@ -62,24 +50,6 @@ If you see ANY `[ ] pending` or `[~] in_progress` tasks, DO NOT provide final an {{ todo_list }} {% endif %} -# MANDATORY Investigation Process - -For ANY question that requires multiple steps or investigation, you MUST follow this structured approach: - -1. **IMMEDIATELY START with TodoWrite**: Before doing any investigation, you MUST call the TodoWrite tool to create an investigation plan. Break down the question into specific tasks. - -2. **Problem Analysis**: Identify what sub-problems need to be solved and create tasks for each. - -3. **Task Execution**: Execute your plan systematically. You MUST: - - Mark tasks as 'in_progress' when you start working on them - - Call TodoWrite to update task status as you complete each task - - Reference the visible task list from your TodoWrite output throughout your investigation - - Follow the exact tasks you created - complete all of them - -4. **Verification**: Before providing your final answer, verify that your conclusion is complete and addresses the user's question fully. - -IMPORTANT: If a question requires more than one investigation step, your FIRST tool call MUST be TodoWrite. - # MANDATORY Multi-Phase Investigation Process For ANY question requiring investigation, you MUST follow this structured approach: @@ -110,20 +80,12 @@ After completing ALL tasks in current list, you MUST: - "I can provide a complete answer with specific, actionable information" - "No additional investigation would significantly improve my answer" -## MANDATORY Final Phase: Verification -When investigation is complete, you MUST create one final verification phase: - -1. **Create Verification Task List** using TodoWrite with these EXACT tasks: - TodoWrite(todos=[ - {"id": "v1", "content": "Verify my answer completely addresses the original user question", "status": "pending"}, - {"id": "v2", "content": "Confirm all claims in my answer are backed by tool output evidence", "status": "pending"}, - {"id": "v3", "content": "Validate root cause analysis follows five whys methodology correctly", "status": "pending"}, - {"id": "v4", "content": "Check that specific actionable information is provided (exact names, commands, steps)", "status": "pending"}, - {"id": "v5", "content": "Ensure no assumptions were made without verification through tools", "status": "pending"}, - {"id": "v6", "content": "Self-critique: identify any weaknesses or gaps in my proposed answer", "status": "pending"} - ]) +## MANDATORY Final Phase: Final Review -2. **Execute ALL verification tasks** systematically before final answer + **Before providing final answer, you MUST:** + - Confirm answer addresses user question completely + - Verify all claims backed by tool evidence + - Ensure actionable information provided ## CRITICAL ENFORCEMENT RULES From fc9a805030ba1bcfc53a668ff1f47354b1878eab Mon Sep 17 00:00:00 2001 From: Arik Alon Date: Tue, 12 Aug 2025 22:37:48 +0300 Subject: [PATCH 07/17] test fix --- .../test_ask_holmes/63_fetch_error_logs_no_errors/test_case.yaml | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/llm/fixtures/test_ask_holmes/63_fetch_error_logs_no_errors/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/63_fetch_error_logs_no_errors/test_case.yaml index 4027bb9005..08d79a954e 100644 --- a/tests/llm/fixtures/test_ask_holmes/63_fetch_error_logs_no_errors/test_case.yaml +++ b/tests/llm/fixtures/test_ask_holmes/63_fetch_error_logs_no_errors/test_case.yaml @@ -5,6 +5,7 @@ before_test: | kubectl create namespace staging-63 || true kubectl create secret generic app-code-63 -n staging-63 --from-file=app.py=./app.py --dry-run=client -o yaml | kubectl apply -f - kubectl apply -f ./manifest.yaml + kubectl wait --for=condition=available deployment/user-api -n staging-63 --timeout=120s after_test: | kubectl delete -f ./manifest.yaml kubectl delete secret app-code-63 -n staging-63 --ignore-not-found From 791b647035fa7016267110f6d25730c217123104 Mon Sep 17 00:00:00 2001 From: Arik Alon Date: Wed, 13 Aug 2025 14:47:05 +0300 Subject: [PATCH 08/17] test fix --- holmes/core/tools.py | 1 - holmes/plugins/prompts/generic_ask.jinja2 | 6 +- .../prompts/investigation_procedure.jinja2 | 94 +++++++++++++------ 3 files changed, 66 insertions(+), 35 deletions(-) diff --git a/holmes/core/tools.py b/holmes/core/tools.py index f99b86ca3b..06bdc4f2a8 100644 --- a/holmes/core/tools.py +++ b/holmes/core/tools.py @@ -608,7 +608,6 @@ def _invoke(self, params: Dict) -> StructuredToolResult: ) try: - logging.info(f"### invoking todowrite tool {params}") todos_data = params.get("todos", []) tasks = [] diff --git a/holmes/plugins/prompts/generic_ask.jinja2 b/holmes/plugins/prompts/generic_ask.jinja2 index bf5af2e130..f5679319be 100644 --- a/holmes/plugins/prompts/generic_ask.jinja2 +++ b/holmes/plugins/prompts/generic_ask.jinja2 @@ -5,15 +5,15 @@ Do not say 'based on the tool output' or explicitly refer to tools at all. If you output an answer and then realize you need to call more tools or there are possible next steps, you may do so by calling tools at that point in time. If you have a good and concrete suggestion for how the user can fix something, tell them even if not asked explicitly +If you are unsure about the answer to the user's request or how to satisfy their request, you should gather more information. This can be done by asking the user for more information. +Bias towards not asking the user for help if you can find the answer yourself. + {% include 'investigation_procedure.jinja2' %} {% include '_current_date_time.jinja2' %} Use conversation history to maintain continuity when appropriate, ensuring efficiency in your responses. -If you are unsure about the answer to the user's request or how to satisfy their request, you should gather more information. This can be done by asking the user for more information. -Bias towards not asking the user for help if you can find the answer yourself. - {% include '_general_instructions.jinja2' %} {% include '_runbook_instructions.jinja2' %} diff --git a/holmes/plugins/prompts/investigation_procedure.jinja2 b/holmes/plugins/prompts/investigation_procedure.jinja2 index e369c860bb..a2c55d51c0 100644 --- a/holmes/plugins/prompts/investigation_procedure.jinja2 +++ b/holmes/plugins/prompts/investigation_procedure.jinja2 @@ -6,8 +6,39 @@ CRITICAL: For multi-step questions, you MUST start by calling the TodoWrite tool MANDATORY Task Status Updates: - When starting a task: Call TodoWrite changing that task's status to "in_progress" - When completing a task: Call TodoWrite changing that task's status to "completed" -- Keep all other tasks unchanged in each TodoWrite call -- Only work on ONE task at a time (only one "in_progress" task) + +PARALLEL EXECUTION RULES: +- When possible, work on multiple tasks at a time. If tasks depend on one another, do them one after the other. +- You MAY execute multiple INDEPENDENT tasks simultaneously +- Mark multiple tasks as "in_progress" if they don't depend on each other +- Wait for dependent tasks to complete before starting tasks that need their results +- Always use a single TodoWrite call to update multiple task statuses + +DEPENDENCY ANALYSIS: +Before marking tasks as "in_progress", determine if they are: +- ✅ INDEPENDENT: Can run simultaneously (e.g., "Check pod A logs" + "Check pod B logs") +- ❌ DEPENDENT: One needs results from another (e.g., "Find pod name" → "Get pod logs") + +PARALLEL EXECUTION EXAMPLE: +TodoWrite(todos=[ +{"id": "1", "content": "Check frontend pod logs", "status": "in_progress"}, +{"id": "2", "content": "Check backend service config", "status": "in_progress"}, +{"id": "3", "content": "Analyze network policies", "status": "in_progress"}, +{"id": "4", "content": "Compare logs from both pods", "status": "pending"} # Depends on 1,2 +]) + + +Examples: +- Task 1: find the pod name + Task 2: get the pod logs +Execution Order: Perform Task 2 after Task 1 +- Task 1: get the pod events + Task 2: get the pod logs +Execution Order: Perform both tasks together + +MAXIMIZE PARALLEL TOOL CALLS: +- When executing multiple in_progress tasks, make ALL their tool calls at once +- Example: If tasks 1,2,3 are in_progress, call kubectl_logs + kubectl_describe + kubectl_get simultaneously # CRITICAL: TASK COMPLETION ENFORCEMENT @@ -67,6 +98,9 @@ After completing ALL tasks in current list, you MUST: - "Are there gaps, unexplored areas, or additional root causes to investigate?" - "Have I followed the 'five whys' methodology to the actual root cause?" - "Did my investigation reveal new questions or areas that need exploration?" + - "Are there any additional investigation steps I can perform, in order to provide a more accurate solution?" + +If the answer to any of those questions is 'yes' - The investigation is INCOMPLETE! 2. **If Investigation is INCOMPLETE**: - Call TodoWrite to create a NEW task list for the next investigation phase @@ -76,68 +110,66 @@ After completing ALL tasks in current list, you MUST: - Repeat this evaluation process 3. **Continue Creating New Phases** until you can answer "YES" to: + - "Do I have enough information to completely answer the user's question?" + - "Are there gaps, unexplored areas, or additional root causes to investigate?" + - "Have I followed the 'five whys' methodology to the actual root cause?" + - "Did my investigation reveal new questions or areas that need exploration?" + - "Are there any additional investigation steps I can perform, in order to provide a more accurate solution?" - "I have thoroughly investigated all aspects of this problem" - "I can provide a complete answer with specific, actionable information" - - "No additional investigation would significantly improve my answer" + - "No additional investigation would improve my answer" ## MANDATORY Final Phase: Final Review **Before providing final answer, you MUST:** - - Confirm answer addresses user question completely + - Confirm answer addresses user question completely! This is the most important thing - Verify all claims backed by tool evidence - Ensure actionable information provided ## CRITICAL ENFORCEMENT RULES **ABSOLUTE REQUIREMENTS:** - - NO final answer until verification phase is 100% completed + - NO final answer until the final review phase is 100% completed - Each investigation phase must have ALL tasks completed before evaluation - You MUST explicitly create new investigation phases when gaps are identified - - Verification phase is MANDATORY - never skip it - - Mark each task as "in_progress" when working on it, "completed" when done + - Final Review phase is MANDATORY - never skip it **EXAMPLES of Phase Progression:** *Phase 1*: Initial investigation discovers pod crashes *Phase 2*: Deep dive into specific pod logs and resource constraints *Phase 3*: Investigate upstream services causing the crashes - *Verification Phase*: Self-critique and validate the complete solution + *Final Review Phase*: Self-critique and validate the complete solution + + *Phase 1*: Initial investigation - check pod health, metrics, logs, traces + *Phase 2*: Based on data from the traces in Phase 1, investigate another workload in the cluster, that seem to be the root cause of the issue. Investigate this workload as well + *Phase 3*: Based on logs gathered in Phase 2, investigate a 3rd party managed service, that seems to be the cause for the whole chain of events. + *Final Review Phase*: Validate that the chain of events, accross the different components, can lead to the investigated scenario. **VIOLATION CONSEQUENCES:** - - Providing answers without verification phase = INVESTIGATION FAILURE + - Providing answers without Final Review phase = INVESTIGATION FAILURE - Skipping investigation phases when gaps exist = INCOMPLETE ANALYSIS - Not completing all tasks in a phase = PROCESS VIOLATION -# VERIFICATION PHASE EXECUTION GUIDE +# FINAL REVIEW PHASE EXECUTION GUIDE - When executing verification tasks, you must: - - **For Task v1 (Answer Completeness):** + When executing Final Review, you must: - Reread the original user question word-by-word - Compare against your proposed answer - Identify any aspects not addressed - - **For Task v2 (Evidence Backing):** + - Make sure you answer what the user asked! - List each claim in your answer - Trace each claim back to specific tool outputs - Flag any unsupported statements - - **For Task v3 (Root Cause Analysis):** - Walk through your "five whys" chain - Verify each "why" logically follows from evidence - Ensure you reached actual root cause, not just symptoms - - **For Task v4 (Actionable Information):** - Verify exact resource names are provided (not generic examples) - Check commands are complete and runnable - Ensure steps are specific to user's environment - - **For Task v5 (No Assumptions):** - List any resource names, namespaces, configurations mentioned - Verify each was confirmed via tool calls - Flag anything assumed without verification - - **For Task v6 (Self-Critique):** - Identify potential weaknesses in your investigation - Consider alternative explanations not explored - Assess if additional investigation would strengthen answer @@ -145,14 +177,14 @@ After completing ALL tasks in current list, you MUST: # INVESTIGATION PHASE TRANSITION EXAMPLES - **Example 1: Network Connectivity Issue** - Phase 1: Check pod status, basic connectivity - → Evaluation: Found network issues, but need to investigate underlying cause - Phase 2: Investigate network policies, DNS resolution - → Evaluation: Found DNS issues, need to check DNS server configuration - Phase 3: Deep dive into DNS server pods, configuration - → Evaluation: Complete - found misconfigured DNS upstream - Verification Phase: Validate solution addresses original connectivity problem + **Example 1: Increased Error Rate** + Phase 1: Check pod status, basic connectivity, logs, traces + → Evaluation: From traces, detected that the error is related to an upstream service + Phase 2: Investigate the upstream service detected in Phase 1 + → Evaluation: Found the upstream service has error while connecting to a managed storage service. + Phase 3: Investigate the external managed storage found in Phase 2 + → Evaluation: Complete - found managed service is down due to outage + Verification Phase: Validate solution addresses original increased error rate. **Example 2: Application Performance Issue** Phase 1: Check application metrics, resource usage From 86919c09f3d6375aee0472e71a575fe8aecd7445 Mon Sep 17 00:00:00 2001 From: Arik Alon Date: Wed, 13 Aug 2025 19:46:53 +0300 Subject: [PATCH 09/17] remove debug info --- holmes/core/tools.py | 1 - 1 file changed, 1 deletion(-) diff --git a/holmes/core/tools.py b/holmes/core/tools.py index f99b86ca3b..06bdc4f2a8 100644 --- a/holmes/core/tools.py +++ b/holmes/core/tools.py @@ -608,7 +608,6 @@ def _invoke(self, params: Dict) -> StructuredToolResult: ) try: - logging.info(f"### invoking todowrite tool {params}") todos_data = params.get("todos", []) tasks = [] From e3a637b14c77fe0ecd123a7f7f4a1e2ce6b41df5 Mon Sep 17 00:00:00 2001 From: Arik Alon Date: Sun, 17 Aug 2025 18:27:38 +0300 Subject: [PATCH 10/17] organize the code --- docs/installation/python-installation.md | 3 + holmes/core/conversations.py | 11 ++ holmes/core/investigation.py | 8 +- holmes/core/prompt.py | 2 + holmes/core/todo_manager.py | 37 +---- holmes/core/tool_calling_llm.py | 17 ++- holmes/core/tools.py | 131 ------------------ holmes/interactive.py | 1 + holmes/main.py | 1 + .../prompts/_general_instructions.jinja2 | 2 + .../prompts/investigation_procedure.jinja2 | 5 + .../prompts/kubernetes_workload_ask.jinja2 | 2 + .../investigator/core_investigation.py | 122 +++++++++++++++- holmes/plugins/toolsets/investigator/model.py | 15 ++ server.py | 1 + tests/core/test_prompt.py | 7 + tests/core/test_todo_manager.py | 12 +- tests/core/test_todo_write_tool.py | 63 +-------- tests/llm/test_ask_holmes.py | 1 + 19 files changed, 186 insertions(+), 255 deletions(-) create mode 100644 holmes/plugins/toolsets/investigator/model.py diff --git a/docs/installation/python-installation.md b/docs/installation/python-installation.md index e3ada2e0c1..22793251d3 100644 --- a/docs/installation/python-installation.md +++ b/docs/installation/python-installation.md @@ -48,6 +48,7 @@ messages = build_initial_ask_messages( initial_user_prompt=question, file_paths=None, tool_executor=ai.tool_executor, + investigation_id=ai.investigation_id, runbooks=config.get_runbook_catalog(), system_prompt_additions=None ) @@ -129,6 +130,7 @@ def main(): initial_user_prompt=question, file_paths=None, tool_executor=ai.tool_executor, + investigation_id=ai.investigation_id, runbooks=config.get_runbook_catalog(), system_prompt_additions=None ) @@ -222,6 +224,7 @@ def main(): initial_user_prompt=first_question, file_paths=None, tool_executor=ai.tool_executor, + investigation_id=ai.investigation_id, runbooks=config.get_runbook_catalog(), system_prompt_additions=None ) diff --git a/holmes/core/conversations.py b/holmes/core/conversations.py index c61cc612a6..9a306f2160 100644 --- a/holmes/core/conversations.py +++ b/holmes/core/conversations.py @@ -133,6 +133,7 @@ def build_issue_chat_messages( "issue": issue_chat_request.issue_type, "toolsets": ai.tool_executor.toolsets, "cluster_name": config.cluster_name, + "investigation_id": ai.investigation_id, }, ) messages = [ @@ -153,6 +154,7 @@ def build_issue_chat_messages( "issue": issue_chat_request.issue_type, "toolsets": ai.tool_executor.toolsets, "cluster_name": config.cluster_name, + "investigation_id": ai.investigation_id, } system_prompt_without_tools = load_and_render_prompt( template_path, template_context_without_tools @@ -186,6 +188,7 @@ def build_issue_chat_messages( "issue": issue_chat_request.issue_type, "toolsets": ai.tool_executor.toolsets, "cluster_name": config.cluster_name, + "investigation_id": ai.investigation_id, } system_prompt_with_truncated_tools = load_and_render_prompt( template_path, truncated_template_context @@ -227,6 +230,7 @@ def build_issue_chat_messages( "issue": issue_chat_request.issue_type, "toolsets": ai.tool_executor.toolsets, "cluster_name": config.cluster_name, + "investigation_id": ai.investigation_id, } system_prompt_without_tools = load_and_render_prompt( template_path, template_context_without_tools @@ -250,6 +254,7 @@ def build_issue_chat_messages( "issue": issue_chat_request.issue_type, "toolsets": ai.tool_executor.toolsets, "cluster_name": config.cluster_name, + "investigation_id": ai.investigation_id, } system_prompt_with_truncated_tools = load_and_render_prompt( template_path, template_context @@ -274,6 +279,7 @@ def add_or_update_system_prompt( context = { "toolsets": ai.tool_executor.toolsets, "cluster_name": config.cluster_name, + "investigation_id": ai.investigation_id, } system_prompt = load_and_render_prompt(template_path, context) @@ -465,6 +471,7 @@ def build_workload_health_chat_messages( "resource": resource, "toolsets": ai.tool_executor.toolsets, "cluster_name": config.cluster_name, + "investigation_id": ai.investigation_id, }, ) messages = [ @@ -485,6 +492,7 @@ def build_workload_health_chat_messages( "resource": resource, "toolsets": ai.tool_executor.toolsets, "cluster_name": config.cluster_name, + "investigation_id": ai.investigation_id, } system_prompt_without_tools = load_and_render_prompt( template_path, template_context_without_tools @@ -518,6 +526,7 @@ def build_workload_health_chat_messages( "resource": resource, "toolsets": ai.tool_executor.toolsets, "cluster_name": config.cluster_name, + "investigation_id": ai.investigation_id, } system_prompt_with_truncated_tools = load_and_render_prompt( template_path, truncated_template_context @@ -559,6 +568,7 @@ def build_workload_health_chat_messages( "resource": resource, "toolsets": ai.tool_executor.toolsets, "cluster_name": config.cluster_name, + "investigation_id": ai.investigation_id, } system_prompt_without_tools = load_and_render_prompt( template_path, template_context_without_tools @@ -582,6 +592,7 @@ def build_workload_health_chat_messages( "resource": resource, "toolsets": ai.tool_executor.toolsets, "cluster_name": config.cluster_name, + "investigation_id": ai.investigation_id, } system_prompt_with_truncated_tools = load_and_render_prompt( template_path, template_context diff --git a/holmes/core/investigation.py b/holmes/core/investigation.py index 918e01537d..f44be456a5 100644 --- a/holmes/core/investigation.py +++ b/holmes/core/investigation.py @@ -10,6 +10,7 @@ from holmes.core.tracing import DummySpan, SpanType from holmes.utils.global_instructions import add_global_instructions_to_user_prompt from holmes.utils.robusta import load_robusta_api_key +from holmes.core.todo_manager import get_todo_manager from holmes.core.investigation_structured_output import ( DEFAULT_SECTIONS, @@ -133,12 +134,8 @@ def get_investigation_context( else: logging.info("Structured output is disabled for this request") - # Add TodoList context to prompt - from holmes.core.todo_manager import get_todo_manager, get_session_id_from_context - todo_manager = get_todo_manager() - session_id = get_session_id_from_context() - todo_context = todo_manager.format_tasks_for_prompt(session_id) + todo_context = todo_manager.format_tasks_for_prompt(ai.investigation_id) system_prompt = load_and_render_prompt( investigate_request.prompt_template, @@ -149,6 +146,7 @@ def get_investigation_context( "toolsets": ai.tool_executor.toolsets, "cluster_name": config.cluster_name, "todo_list": todo_context, + "investigation_id": ai.investigation_id, }, ) diff --git a/holmes/core/prompt.py b/holmes/core/prompt.py index c317658e8d..477a6d0a70 100644 --- a/holmes/core/prompt.py +++ b/holmes/core/prompt.py @@ -30,6 +30,7 @@ def build_initial_ask_messages( initial_user_prompt: str, file_paths: Optional[List[Path]], tool_executor: Any, # ToolExecutor type + investigation_id: str, runbooks: Union[RunbookCatalog, Dict, None] = None, system_prompt_additions: Optional[str] = None, ) -> List[Dict]: @@ -49,6 +50,7 @@ def build_initial_ask_messages( "toolsets": tool_executor.toolsets, "runbooks": runbooks or {}, "system_prompt_additions": system_prompt_additions or "", + "investigation_id": investigation_id, } system_prompt_rendered = load_and_render_prompt( system_prompt_template, template_context diff --git a/holmes/core/todo_manager.py b/holmes/core/todo_manager.py index ba661fc0a1..1aaf4aeb2f 100644 --- a/holmes/core/todo_manager.py +++ b/holmes/core/todo_manager.py @@ -1,9 +1,7 @@ -import logging from typing import Dict, List from threading import Lock -from uuid import uuid4 -from holmes.core.tools import Task, TaskStatus +from holmes.plugins.toolsets.investigator.model import Task, TaskStatus class TodoListManager: @@ -17,24 +15,19 @@ def __init__(self): self._lock = Lock() def get_session_tasks(self, session_id: str) -> List[Task]: - """Get all tasks for a session. Returns empty list if session doesn't exist.""" - logging.info(f"########## get_session_tasks {session_id}") with self._lock: return self._sessions.get(session_id, []).copy() def update_session_tasks(self, session_id: str, tasks: List[Task]) -> None: - """Update all tasks for a session.""" with self._lock: self._sessions[session_id] = tasks.copy() def clear_session(self, session_id: str) -> None: - """Clear all tasks for a session.""" with self._lock: if session_id in self._sessions: del self._sessions[session_id] def get_session_count(self) -> int: - """Get number of active sessions (for debugging/testing).""" with self._lock: return len(self._sessions) @@ -48,7 +41,6 @@ def format_tasks_for_prompt(self, session_id: str) -> str: if not tasks: return "" - # Sort tasks by status (pending -> in_progress -> completed) then priority status_order = {"pending": 0, "in_progress": 1, "completed": 2} sorted_tasks = sorted( @@ -59,7 +51,6 @@ def format_tasks_for_prompt(self, session_id: str) -> str: lines = ["# CURRENT INVESTIGATION TASKS"] lines.append("") - # Count tasks by status pending_count = sum(1 for t in tasks if t.status == TaskStatus.PENDING) progress_count = sum(1 for t in tasks if t.status == TaskStatus.IN_PROGRESS) completed_count = sum(1 for t in tasks if t.status == TaskStatus.COMPLETED) @@ -70,7 +61,6 @@ def format_tasks_for_prompt(self, session_id: str) -> str: lines.append("") for task in sorted_tasks: - # Use simple text indicators for prompt injection status_indicator = { "pending": "[ ]", "in_progress": "[~]", @@ -87,33 +77,8 @@ def format_tasks_for_prompt(self, session_id: str) -> str: return "\n".join(lines) -# Global instance for session management _todo_manager = TodoListManager() def get_todo_manager() -> TodoListManager: - """Get the global TodoListManager instance.""" return _todo_manager - - -def get_session_id_from_context(context=None) -> str: - """ - Extract or generate session ID from context. - For now, we'll use a simple approach - this can be enhanced later. - """ - # TODO: This should be enhanced to extract session ID from investigation context - # For now, we'll use a simple thread-local or context-based approach - if hasattr(context, "_current_session"): - return context._current_session - - # Generate new session ID - session_id = str(uuid4()) - if context: - context._current_session = session_id - return session_id - - -def set_current_session_id(session_id: str) -> None: - """Set the current session ID for this thread/context.""" - pass - # get_session_id_from_context._current_session = session_id diff --git a/holmes/core/tool_calling_llm.py b/holmes/core/tool_calling_llm.py index 3944923fef..6b66ee8dbd 100644 --- a/holmes/core/tool_calling_llm.py +++ b/holmes/core/tool_calling_llm.py @@ -2,6 +2,7 @@ import json import logging import textwrap +import uuid from typing import Dict, List, Optional, Type, Union import sentry_sdk @@ -38,6 +39,9 @@ from holmes.core.tracing import DummySpan from holmes.utils.colors import AI_COLOR from holmes.utils.stream import StreamEvents, StreamMessage +from holmes.core.todo_manager import ( + get_todo_manager, +) def format_tool_result_data(tool_result: StructuredToolResult) -> str: @@ -204,9 +208,10 @@ def __init__( self, tool_executor: ToolExecutor, max_steps: int, llm: LLM, tracer=None ): self.tool_executor = tool_executor - self.max_steps = 40 ## TODO Arik - remove this + self.max_steps = max_steps self.tracer = tracer self.llm = llm + self.investigation_id = str(uuid.uuid4()) def prompt_call( self, @@ -776,15 +781,8 @@ def investigate( "[bold]No runbooks found for this issue. Using default behaviour. (Add runbooks to guide the investigation.)[/bold]" ) - # Add TodoList context to prompt - from holmes.core.todo_manager import ( - get_todo_manager, - get_session_id_from_context, - ) - todo_manager = get_todo_manager() - session_id = get_session_id_from_context() - todo_context = todo_manager.format_tasks_for_prompt(session_id) + todo_context = todo_manager.format_tasks_for_prompt(self.investigation_id) system_prompt = load_and_render_prompt( prompt, @@ -795,6 +793,7 @@ def investigate( "toolsets": self.tool_executor.toolsets, "cluster_name": self.cluster_name, "todo_list": todo_context, + "investigation_id": self.investigation_id, }, ) diff --git a/holmes/core/tools.py b/holmes/core/tools.py index 06bdc4f2a8..cff65e18f0 100644 --- a/holmes/core/tools.py +++ b/holmes/core/tools.py @@ -18,7 +18,6 @@ from holmes.plugins.prompts import load_and_render_prompt import time from rich.table import Table -from uuid import uuid4 class ToolResultStatus(str, Enum): @@ -570,133 +569,3 @@ def pretty_print_toolset_status(toolsets: list[Toolset], console: Console) -> No table.add_row(*(str(row.get(col.capitalize(), "")) for col in status_fields)) console.print(table) - - -class TaskStatus(str, Enum): - PENDING = "pending" - IN_PROGRESS = "in_progress" - COMPLETED = "completed" - - -class TaskPriority(str, Enum): - HIGH = "high" - MEDIUM = "medium" - LOW = "low" - - -class Task(BaseModel): - id: str = Field(default_factory=lambda: str(uuid4())) - content: str - status: TaskStatus = TaskStatus.PENDING - - -class TodoWriteTool(Tool): - name: str = "TodoWrite" - description: str = "Save investigation tasks to break down complex problems into manageable sub-tasks" - parameters: Dict[str, ToolParameter] = { - "todos": ToolParameter( - description="List of tasks to track during the investigation. Each task should have: id (string), content (string), status (pending/in_progress/completed)", - type="array[object]", - required=True, - ) - } - - def _invoke(self, params: Dict) -> StructuredToolResult: - from holmes.core.todo_manager import ( - get_todo_manager, - get_session_id_from_context, - ) - - try: - todos_data = params.get("todos", []) - - tasks = [] - - for todo_item in todos_data: - if isinstance(todo_item, dict): - task = Task( - id=todo_item.get("id", str(uuid4())), - content=todo_item.get("content", ""), - status=TaskStatus(todo_item.get("status", "pending")), - ) - tasks.append(task) - - logging.info(f"Tasks: {len(tasks)}") - - # Store tasks in session storage - todo_manager = get_todo_manager() - session_id = get_session_id_from_context() - todo_manager.update_session_tasks(session_id, tasks) - - # Print a nice table to console/log - def print_tasks_table(tasks): - if not tasks: - logging.info("No tasks in the investigation plan.") - return - - # Calculate column widths - max_id_width = max(len(str(task.id)) for task in tasks) - max_content_width = max(len(task.content) for task in tasks) - max_status_width = max(len(task.status.value) for task in tasks) - - # Ensure minimum widths for headers - id_width = max(max_id_width, 2) - content_width = max(max_content_width, 7) - status_width = max(max_status_width, 6) - - # Status indicators - status_icons = { - "pending": "[ ]", - "in_progress": "[~]", - "completed": "[✓]", - } - - # Build table - separator = f"+{'-' * (id_width + 2)}+{'-' * (content_width + 2)}+{'-' * (status_width + 2)}+" - header = f"| {'ID':<{id_width}} | {'Content':<{content_width}} | {'Status':<{status_width}} |" - - # Log the table - logging.info("Updated Investigation Tasks:") - logging.info(separator) - logging.info(header) - logging.info(separator) - - for task in tasks: - status_display = ( - f"{status_icons[task.status.value]} {task.status.value}" - ) - row = f"| {task.id:<{id_width}} | {task.content:<{content_width}} | {status_display:<{status_width}} |" - logging.info(row) - - logging.info(separator) - - # Print table to console/log - print_tasks_table(tasks) - - # Get pretty formatted version of the updated TodoList (keep existing format) - formatted_tasks = todo_manager.format_tasks_for_prompt(session_id) - - # Return confirmation with pretty printed TodoList (unchanged) - response_data = f"✅ Investigation plan updated with {len(tasks)} tasks. Tasks are now stored in session and will appear in subsequent prompts.\n\n" - if formatted_tasks: - response_data += formatted_tasks - else: - response_data += "No tasks currently in the investigation plan." - - return StructuredToolResult( - status=ToolResultStatus.SUCCESS, - data=response_data, - params=params, - ) - - except Exception as e: - logging.exception("error using todowrite tool") - return StructuredToolResult( - status=ToolResultStatus.ERROR, - error=f"Failed to process tasks: {str(e)}", - params=params, - ) - - def get_parameterized_one_liner(self, params: Dict) -> str: - todos = params.get("todos", []) - return f"Write {todos} investigation tasks" diff --git a/holmes/interactive.py b/holmes/interactive.py index 0c6d6c5d38..4a98f7f014 100644 --- a/holmes/interactive.py +++ b/holmes/interactive.py @@ -1002,6 +1002,7 @@ def get_bottom_toolbar(): user_input, include_files, ai.tool_executor, + ai.investigation_id, runbooks, system_prompt_additions, ) diff --git a/holmes/main.py b/holmes/main.py index 9f8adecf25..ba4639c5b7 100644 --- a/holmes/main.py +++ b/holmes/main.py @@ -302,6 +302,7 @@ def ask( prompt, # type: ignore include_file, ai.tool_executor, + ai.investigation_id, config.get_runbook_catalog(), system_prompt_additions, ) diff --git a/holmes/plugins/prompts/_general_instructions.jinja2 b/holmes/plugins/prompts/_general_instructions.jinja2 index a8e5b8d053..ab060615ad 100644 --- a/holmes/plugins/prompts/_general_instructions.jinja2 +++ b/holmes/plugins/prompts/_general_instructions.jinja2 @@ -1,3 +1,5 @@ +{% include 'investigation_procedure.jinja2' %} + # In general {% if cluster_name -%} diff --git a/holmes/plugins/prompts/investigation_procedure.jinja2 b/holmes/plugins/prompts/investigation_procedure.jinja2 index a2c55d51c0..a0d63d3773 100644 --- a/holmes/plugins/prompts/investigation_procedure.jinja2 +++ b/holmes/plugins/prompts/investigation_procedure.jinja2 @@ -1,3 +1,8 @@ +{% if investigation_id %} +# Investigation ID for this session +Investigation id: {{ investigation_id }} +{% endif %} + CRITICAL: For multi-step questions, you MUST start by calling the TodoWrite tool with a `todos` parameter containing an array of task objects. Each task must have: - `id`: unique identifier (string) - `content`: specific task description (string) diff --git a/holmes/plugins/prompts/kubernetes_workload_ask.jinja2 b/holmes/plugins/prompts/kubernetes_workload_ask.jinja2 index 35f2483174..90eb752a68 100644 --- a/holmes/plugins/prompts/kubernetes_workload_ask.jinja2 +++ b/holmes/plugins/prompts/kubernetes_workload_ask.jinja2 @@ -6,6 +6,8 @@ If you output an answer and then realize you need to call more tools or there ar If the user provides you with extra instructions in a triple single quotes section, ALWAYS perform their instructions and then perform your investigation. {% include '_current_date_time.jinja2' %} +{% include 'investigation_procedure.jinja2' %} + Global Instructions You may receive a set of “Global Instructions” that describe how to perform certain tasks, handle certain situations, or apply certain best practices. They are not mandatory for every request, but serve as a reference resource and must be used if the current scenario or user request aligns with one of the described methods or conditions. Use these rules when deciding how to apply them: diff --git a/holmes/plugins/toolsets/investigator/core_investigation.py b/holmes/plugins/toolsets/investigator/core_investigation.py index c3e927f99b..ee84fb929d 100644 --- a/holmes/plugins/toolsets/investigator/core_investigation.py +++ b/holmes/plugins/toolsets/investigator/core_investigation.py @@ -2,7 +2,125 @@ import os from typing import Any, Dict -from holmes.core.tools import TodoWriteTool, Toolset, ToolsetTag, ToolsetStatusEnum +from uuid import uuid4 +from holmes.core.todo_manager import ( + get_todo_manager, +) + +from holmes.core.tools import ( + Toolset, + ToolsetTag, + ToolsetStatusEnum, + ToolParameter, + Tool, + StructuredToolResult, + ToolResultStatus, +) +from holmes.plugins.toolsets.investigator.model import Task, TaskStatus + + +class TodoWriteTool(Tool): + name: str = "TodoWrite" + description: str = "Save investigation tasks to break down complex problems into manageable sub-tasks" + parameters: Dict[str, ToolParameter] = { + "todos": ToolParameter( + description="List of tasks to track during the investigation. Each task should have: id (string), content (string), status (pending/in_progress/completed)", + type="array[object]", + required=True, + ), + "investigation_id": ToolParameter( + description="This investigation identifier. This is a uuid that represents the investigation session id.", + type="string", + required=True, + ), + } + + # Print a nice table to console/log + def print_tasks_table(self, tasks): + if not tasks: + logging.info("No tasks in the investigation plan.") + return + + max_id_width = max(len(str(task.id)) for task in tasks) + max_content_width = max(len(task.content) for task in tasks) + max_status_width = max(len(task.status.value) for task in tasks) + + id_width = max(max_id_width, 2) + content_width = max(max_content_width, 7) + status_width = max(max_status_width, 6) + + status_icons = { + "pending": "[ ]", + "in_progress": "[~]", + "completed": "[✓]", + } + + # Build table + separator = f"+{'-' * (id_width + 2)}+{'-' * (content_width + 2)}+{'-' * (status_width + 2)}+" + header = f"| {'ID':<{id_width}} | {'Content':<{content_width}} | {'Status':<{status_width}} |" + + # Log the table + logging.info("Updated Investigation Tasks:") + logging.info(separator) + logging.info(header) + logging.info(separator) + + for task in tasks: + status_display = f"{status_icons[task.status.value]} {task.status.value}" + row = f"| {task.id:<{id_width}} | {task.content:<{content_width}} | {status_display:<{status_width}} |" + logging.info(row) + + logging.info(separator) + + def _invoke(self, params: Dict) -> StructuredToolResult: + try: + todos_data = params.get("todos", []) + + tasks = [] + + for todo_item in todos_data: + if isinstance(todo_item, dict): + task = Task( + id=todo_item.get("id", str(uuid4())), + content=todo_item.get("content", ""), + status=TaskStatus(todo_item.get("status", "pending")), + ) + tasks.append(task) + + logging.info(f"Tasks: {len(tasks)}") + + # Store tasks in session storage + todo_manager = get_todo_manager() + session_id = params.get("investigation_id", "") + todo_manager.update_session_tasks(session_id, tasks) + + self.print_tasks_table(tasks) + + formatted_tasks = todo_manager.format_tasks_for_prompt(session_id) + + response_data = f"✅ Investigation plan updated with {len(tasks)} tasks. Tasks are now stored in session and will appear in subsequent prompts.\n\n" + if formatted_tasks: + response_data += formatted_tasks + else: + response_data += "No tasks currently in the investigation plan." + + return StructuredToolResult( + status=ToolResultStatus.SUCCESS, + data=response_data, + params=params, + ) + + except Exception as e: + logging.exception("error using todowrite tool") + return StructuredToolResult( + status=ToolResultStatus.ERROR, + error=f"Failed to process tasks: {str(e)}", + params=params, + ) + + def get_parameterized_one_liner(self, params: Dict) -> str: + todos = params.get("todos", []) + return f"Write {todos} investigation tasks" class CoreInvestigationToolset(Toolset): @@ -17,7 +135,6 @@ def __init__(self): tags=[ToolsetTag.CORE], is_default=True, ) - # Override the default DISABLED status to ENABLED since this is a core toolset self.status = ToolsetStatusEnum.ENABLED logging.info("Core investigation toolset loaded") @@ -25,7 +142,6 @@ def get_example_config(self) -> Dict[str, Any]: return {} def _reload_instructions(self): - """Load Datadog metrics specific troubleshooting instructions.""" template_file_path = os.path.abspath( os.path.join(os.path.dirname(__file__), "investigator_instructions.jinja2") ) diff --git a/holmes/plugins/toolsets/investigator/model.py b/holmes/plugins/toolsets/investigator/model.py new file mode 100644 index 0000000000..b64277fe86 --- /dev/null +++ b/holmes/plugins/toolsets/investigator/model.py @@ -0,0 +1,15 @@ +from enum import Enum +from pydantic import BaseModel, Field +from uuid import uuid4 + + +class TaskStatus(str, Enum): + PENDING = "pending" + IN_PROGRESS = "in_progress" + COMPLETED = "completed" + + +class Task(BaseModel): + id: str = Field(default_factory=lambda: str(uuid4())) + content: str + status: TaskStatus = TaskStatus.PENDING diff --git a/server.py b/server.py index c160122acb..b9cab9d883 100644 --- a/server.py +++ b/server.py @@ -216,6 +216,7 @@ def workload_health_check(request: WorkloadHealthRequest): "toolsets": ai.tool_executor.toolsets, "response_format": workload_health_structured_output, "cluster_name": config.cluster_name, + "investigation_id": ai.investigation_id, }, ) diff --git a/tests/core/test_prompt.py b/tests/core/test_prompt.py index 4e22096057..cc0892a5e4 100644 --- a/tests/core/test_prompt.py +++ b/tests/core/test_prompt.py @@ -1,3 +1,5 @@ +import uuid + import pytest from unittest.mock import Mock from rich.console import Console @@ -28,6 +30,7 @@ def test_build_initial_ask_messages_basic(console, mock_tool_executor): "Test prompt", None, mock_tool_executor, + str(uuid.uuid4()), None, None, ) @@ -48,6 +51,7 @@ def test_build_initial_ask_messages_with_system_prompt_additions( "Test prompt", None, mock_tool_executor, + str(uuid.uuid4()), None, system_additions, ) @@ -71,6 +75,7 @@ def test_build_initial_ask_messages_with_file(console, mock_tool_executor, tmp_p "Test prompt", [test_file], mock_tool_executor, + str(uuid.uuid4()), None, None, ) @@ -95,6 +100,7 @@ def test_build_initial_ask_messages_with_runbooks(console, mock_tool_executor): "Test prompt", None, mock_tool_executor, + str(uuid.uuid4()), runbooks, None, ) @@ -122,6 +128,7 @@ def test_build_initial_ask_messages_all_parameters( "Test prompt", [test_file], mock_tool_executor, + str(uuid.uuid4()), runbooks, system_additions, ) diff --git a/tests/core/test_todo_manager.py b/tests/core/test_todo_manager.py index a05f5aa1d0..e0b33dc7d5 100644 --- a/tests/core/test_todo_manager.py +++ b/tests/core/test_todo_manager.py @@ -1,10 +1,8 @@ from holmes.core.todo_manager import ( TodoListManager, get_todo_manager, - set_current_session_id, - get_session_id_from_context, ) -from holmes.core.tools import Task, TaskStatus +from holmes.plugins.toolsets.investigator.model import Task, TaskStatus class TestTodoListManager: @@ -98,11 +96,3 @@ def test_global_manager_instance(self): manager1 = get_todo_manager() manager2 = get_todo_manager() assert manager1 is manager2 - - def test_session_id_management(self): - """Test session ID context management.""" - test_session_id = "context-test-session" - set_current_session_id(test_session_id) - - retrieved_session_id = get_session_id_from_context() - assert retrieved_session_id == test_session_id diff --git a/tests/core/test_todo_write_tool.py b/tests/core/test_todo_write_tool.py index 8e3b8c75cf..940b320f12 100644 --- a/tests/core/test_todo_write_tool.py +++ b/tests/core/test_todo_write_tool.py @@ -1,4 +1,6 @@ -from holmes.core.tools import TodoWriteTool, ToolResultStatus, TaskStatus, TaskPriority +from holmes.core.tools import ToolResultStatus +from holmes.plugins.toolsets.investigator.core_investigation import TodoWriteTool +from holmes.plugins.toolsets.investigator.model import TaskStatus class TestTodoWriteTool: @@ -100,12 +102,6 @@ def test_task_status_enum(self): assert TaskStatus.IN_PROGRESS == "in_progress" assert TaskStatus.COMPLETED == "completed" - def test_task_priority_enum(self): - """Test TaskPriority enum values.""" - assert TaskPriority.HIGH == "high" - assert TaskPriority.MEDIUM == "medium" - assert TaskPriority.LOW == "low" - def test_openai_format(self): """Test that the tool generates correct OpenAI format.""" tool = TodoWriteTool() @@ -128,56 +124,3 @@ def test_openai_format(self): # Check required fields assert "todos" in params["required"] - - def test_session_storage_functionality(self): - """Test that the tool stores tasks in session storage.""" - from holmes.core.todo_manager import get_todo_manager, set_current_session_id - - tool = TodoWriteTool() - session_id = "test-session-456" - set_current_session_id(session_id) - - params = { - "todos": [ - { - "id": "1", - "content": "Task 1", - "status": "pending", - "priority": "high", - }, - { - "id": "2", - "content": "Task 2", - "status": "completed", - "priority": "low", - }, - ] - } - - result = tool._invoke(params) - - assert result.status == ToolResultStatus.SUCCESS - assert "2 tasks" in result.data - assert "Investigation plan updated" in result.data - - # Check that the pretty printed TodoList is included in the response - assert "CURRENT INVESTIGATION TASKS" in result.data - assert "Task 1" in result.data - assert "Task 2" in result.data - assert "[ ]" in result.data # pending indicator - assert "[✓]" in result.data # completed indicator - assert "(HIGH)" in result.data # priority indicator - assert "(LOW)" in result.data # priority indicator - - # Check session storage - manager = get_todo_manager() - stored_tasks = manager.get_session_tasks(session_id) - assert len(stored_tasks) == 2 - assert stored_tasks[0].content == "Task 1" - assert stored_tasks[1].content == "Task 2" - - # Check prompt formatting matches what's in the response - prompt_context = manager.format_tasks_for_prompt(session_id) - assert ( - prompt_context in result.data - ) # The formatted tasks should be part of the response diff --git a/tests/llm/test_ask_holmes.py b/tests/llm/test_ask_holmes.py index d1e531ebb4..8d42a47c84 100644 --- a/tests/llm/test_ask_holmes.py +++ b/tests/llm/test_ask_holmes.py @@ -322,6 +322,7 @@ def ask_holmes( test_case.user_prompt, None, ai.tool_executor, + ai.investigation_id, runbooks, ) else: From b18329ea9cf8b6c722085a49c3c32eb13f468071 Mon Sep 17 00:00:00 2001 From: Arik Alon Date: Sun, 17 Aug 2025 22:40:33 +0300 Subject: [PATCH 11/17] increase max steps default prompt fix --- examples/custom_llm.py | 2 +- holmes/config.py | 2 +- holmes/plugins/prompts/generic_ask.jinja2 | 2 -- holmes/plugins/prompts/investigation_procedure.jinja2 | 3 +++ tests/llm/test_ask_holmes.py | 2 +- 5 files changed, 6 insertions(+), 5 deletions(-) diff --git a/examples/custom_llm.py b/examples/custom_llm.py index 9cd9b132e4..1bfda0957c 100644 --- a/examples/custom_llm.py +++ b/examples/custom_llm.py @@ -55,7 +55,7 @@ def ask_holmes(): ) tool_executor = ToolExecutor(load_builtin_toolsets()) - ai = ToolCallingLLM(tool_executor, max_steps=10, llm=MyCustomLLM()) + ai = ToolCallingLLM(tool_executor, max_steps=40, llm=MyCustomLLM()) response = ai.prompt_call(system_prompt, prompt) diff --git a/holmes/config.py b/holmes/config.py index 42e7ac7a66..a9075e41e9 100644 --- a/holmes/config.py +++ b/holmes/config.py @@ -72,7 +72,7 @@ class Config(RobustaBaseConfig): None # if None, read from OPENAI_API_KEY or AZURE_OPENAI_ENDPOINT env var ) model: Optional[str] = "gpt-4o" - max_steps: int = 10 + max_steps: int = 40 cluster_name: Optional[str] = None alertmanager_url: Optional[str] = None diff --git a/holmes/plugins/prompts/generic_ask.jinja2 b/holmes/plugins/prompts/generic_ask.jinja2 index f5679319be..6f67709890 100644 --- a/holmes/plugins/prompts/generic_ask.jinja2 +++ b/holmes/plugins/prompts/generic_ask.jinja2 @@ -8,8 +8,6 @@ If you have a good and concrete suggestion for how the user can fix something, t If you are unsure about the answer to the user's request or how to satisfy their request, you should gather more information. This can be done by asking the user for more information. Bias towards not asking the user for help if you can find the answer yourself. -{% include 'investigation_procedure.jinja2' %} - {% include '_current_date_time.jinja2' %} Use conversation history to maintain continuity when appropriate, ensuring efficiency in your responses. diff --git a/holmes/plugins/prompts/investigation_procedure.jinja2 b/holmes/plugins/prompts/investigation_procedure.jinja2 index a0d63d3773..a7d726384b 100644 --- a/holmes/plugins/prompts/investigation_procedure.jinja2 +++ b/holmes/plugins/prompts/investigation_procedure.jinja2 @@ -3,6 +3,9 @@ Investigation id: {{ investigation_id }} {% endif %} +CLARIFICATION REQUIREMENT: Before starting ANY investigation, if the user's question is ambiguous or lacks critical details, you MUST ask for clarification first. Do NOT create TodoWrite tasks for unclear questions. +Only proceed with TodoWrite and investigation AFTER you have clear, specific requirements. + CRITICAL: For multi-step questions, you MUST start by calling the TodoWrite tool with a `todos` parameter containing an array of task objects. Each task must have: - `id`: unique identifier (string) - `content`: specific task description (string) diff --git a/tests/llm/test_ask_holmes.py b/tests/llm/test_ask_holmes.py index 8d42a47c84..061bfc2287 100644 --- a/tests/llm/test_ask_holmes.py +++ b/tests/llm/test_ask_holmes.py @@ -296,7 +296,7 @@ def ask_holmes( ai = ToolCallingLLM( tool_executor=tool_executor, - max_steps=10, + max_steps=40, llm=DefaultLLM(os.environ.get("MODEL", "gpt-4o"), tracer=tracer), ) From 749d9358328c07f16a070350fc62ed04d897e321 Mon Sep 17 00:00:00 2001 From: Arik Alon Date: Mon, 18 Aug 2025 00:12:37 +0300 Subject: [PATCH 12/17] increase max steps default prompt fix --- tests/plugins/toolsets/test_core_investigation.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/tests/plugins/toolsets/test_core_investigation.py b/tests/plugins/toolsets/test_core_investigation.py index 361c96a616..1c8dd95c3d 100644 --- a/tests/plugins/toolsets/test_core_investigation.py +++ b/tests/plugins/toolsets/test_core_investigation.py @@ -1,7 +1,8 @@ from holmes.plugins.toolsets.investigator.core_investigation import ( CoreInvestigationToolset, + TodoWriteTool, ) -from holmes.core.tools import ToolsetStatusEnum, ToolsetTag, TodoWriteTool +from holmes.core.tools import ToolsetStatusEnum, ToolsetTag class TestCoreInvestigationToolset: From 8dc43ba989c11c4bac91393aa3f0e2b9e0505baf Mon Sep 17 00:00:00 2001 From: Arik Alon Date: Mon, 18 Aug 2025 01:59:48 +0300 Subject: [PATCH 13/17] fix parameter types for openai schema conversion improve prompts --- holmes/core/openai_formatting.py | 20 ++++ holmes/core/tool_calling_llm.py | 3 +- holmes/core/tools.py | 2 + .../prompts/generic_investigation.jinja2 | 28 ------ .../prompts/investigation_procedure.jinja2 | 2 + .../investigator/core_investigation.py | 14 ++- .../investigator_instructions.jinja2 | 98 +++++++++---------- 7 files changed, 82 insertions(+), 85 deletions(-) diff --git a/holmes/core/openai_formatting.py b/holmes/core/openai_formatting.py index b484814705..0aee7a5487 100644 --- a/holmes/core/openai_formatting.py +++ b/holmes/core/openai_formatting.py @@ -24,6 +24,26 @@ def type_to_open_ai_schema(param_attributes: Any, strict_mode: bool) -> dict[str type_obj = {"type": "object"} if strict_mode: type_obj["additionalProperties"] = False + + # Use explicit properties if provided + if hasattr(param_attributes, "properties") and param_attributes.properties: + type_obj["properties"] = { + name: type_to_open_ai_schema(prop, strict_mode) + for name, prop in param_attributes.properties.items() + } + if strict_mode: + type_obj["required"] = list(param_attributes.properties.keys()) + + elif param_type == "array": + # Handle arrays with explicit item schemas + if hasattr(param_attributes, "items") and param_attributes.items: + items_schema = type_to_open_ai_schema(param_attributes.items, strict_mode) + type_obj = {"type": "array", "items": items_schema} + else: + # Fallback for arrays without explicit item schema + type_obj = {"type": "array", "items": {"type": "object"}} + if strict_mode: + type_obj["items"]["additionalProperties"] = False else: match = re.match(pattern, param_type) diff --git a/holmes/core/tool_calling_llm.py b/holmes/core/tool_calling_llm.py index 99982515a4..494b7fc6da 100644 --- a/holmes/core/tool_calling_llm.py +++ b/holmes/core/tool_calling_llm.py @@ -265,9 +265,8 @@ def call( # type: ignore perf_timing.measure("get_all_tools_openai_format") max_steps = self.max_steps i = 0 - print(f"\n\n####### max steps: {max_steps} \n\n") + while i < max_steps: - print(f"\n\n####### current step: {i} \n\n") i += 1 perf_timing.measure(f"start iteration {i}") logging.debug(f"running iteration {i}") diff --git a/holmes/core/tools.py b/holmes/core/tools.py index 873e577e19..89b433bb24 100644 --- a/holmes/core/tools.py +++ b/holmes/core/tools.py @@ -122,6 +122,8 @@ class ToolParameter(BaseModel): description: Optional[str] = None type: str = "string" required: bool = True + properties: Optional[Dict[str, "ToolParameter"]] = None # For object types + items: Optional["ToolParameter"] = None # For array item schemas class Tool(ABC, BaseModel): diff --git a/holmes/plugins/prompts/generic_investigation.jinja2 b/holmes/plugins/prompts/generic_investigation.jinja2 index d0ac1195d5..2789932afe 100644 --- a/holmes/plugins/prompts/generic_investigation.jinja2 +++ b/holmes/plugins/prompts/generic_investigation.jinja2 @@ -3,34 +3,6 @@ Whenever possible you MUST first use tools to investigate then answer the questi Ask for multiple tool calls at the same time as it saves time for the user. Do not say 'based on the tool output' -CRITICAL: For ALL investigations, you MUST start by calling the TodoWrite tool to create your investigation plan. - -{% if todo_list %} -{{ todo_list }} -{% endif %} - -# MANDATORY Investigation Process - -You MUST follow this structured approach for ALL investigations. NO EXCEPTIONS: - -1. **IMMEDIATELY START with TodoWrite**: Before doing ANYTHING else, you MUST call the TodoWrite tool to create an investigation plan. Analyze the issue and break it down into specific tasks. - -2. **Problem Analysis**: Ask yourself: "What are the sub-problems I need to solve to answer this question?" Create tasks for each sub-problem. - -3. **Task Execution**: Execute your investigation plan by calling the appropriate tools. You MUST: - - Mark tasks as 'in_progress' when you start working on them - - Call TodoWrite to update task status as you complete each task - - Reference the visible task list from your TodoWrite output throughout your investigation - - Follow the exact tasks you created - don't skip or ignore any - -4. **Verification**: Once you believe you have found the answer, perform a verification step: - - Cross-check your findings with the original alert/issue details - - Ensure your conclusion addresses the root cause, not just symptoms - - Verify that your analysis is consistent with all collected data - - If verification fails, use TodoWrite to add new tasks and return to investigation - -REMEMBER: Your first tool call MUST ALWAYS be TodoWrite to create your investigation plan. - Provide an terse analysis of the following {{ issue.source_type }} alert/issue and why it is firing. * {% include '_current_date_time.jinja2' %} * If the tool requires string format timestamps, query from 'start_timestamp' until 'end_timestamp' diff --git a/holmes/plugins/prompts/investigation_procedure.jinja2 b/holmes/plugins/prompts/investigation_procedure.jinja2 index a7d726384b..e1271bccc6 100644 --- a/holmes/plugins/prompts/investigation_procedure.jinja2 +++ b/holmes/plugins/prompts/investigation_procedure.jinja2 @@ -133,6 +133,7 @@ If the answer to any of those questions is 'yes' - The investigation is INCOMPLE - Confirm answer addresses user question completely! This is the most important thing - Verify all claims backed by tool evidence - Ensure actionable information provided + - If additional investigation steps are required, start a new investigation phase, and create a new task list to gather the missing information. ## CRITICAL ENFORCEMENT RULES @@ -181,6 +182,7 @@ If the answer to any of those questions is 'yes' - The investigation is INCOMPLE - Identify potential weaknesses in your investigation - Consider alternative explanations not explored - Assess if additional investigation would strengthen answer + - If there are additional investigation steps that can help the user, start a new phase, and create a new task list to perform these steps # INVESTIGATION PHASE TRANSITION EXAMPLES diff --git a/holmes/plugins/toolsets/investigator/core_investigation.py b/holmes/plugins/toolsets/investigator/core_investigation.py index ee84fb929d..f6890d2457 100644 --- a/holmes/plugins/toolsets/investigator/core_investigation.py +++ b/holmes/plugins/toolsets/investigator/core_investigation.py @@ -21,12 +21,20 @@ class TodoWriteTool(Tool): name: str = "TodoWrite" - description: str = "Save investigation tasks to break down complex problems into manageable sub-tasks" + description: str = "Save investigation tasks to break down complex problems into manageable sub-tasks. ALWAYS provide the COMPLETE list of all tasks, not just the ones being updated." parameters: Dict[str, ToolParameter] = { "todos": ToolParameter( - description="List of tasks to track during the investigation. Each task should have: id (string), content (string), status (pending/in_progress/completed)", - type="array[object]", + description="COMPLETE list of ALL tasks on the task list. Each task should have: id (string), content (string), status (pending/in_progress/completed)", + type="array", required=True, + items=ToolParameter( + type="object", + properties={ + "id": ToolParameter(type="string", required=True), + "content": ToolParameter(type="string", required=True), + "status": ToolParameter(type="string", required=True), + }, + ), ), "investigation_id": ToolParameter( description="This investigation identifier. This is a uuid that represents the investigation session id.", diff --git a/holmes/plugins/toolsets/investigator/investigator_instructions.jinja2 b/holmes/plugins/toolsets/investigator/investigator_instructions.jinja2 index c8f6da4b10..3f50a1604b 100644 --- a/holmes/plugins/toolsets/investigator/investigator_instructions.jinja2 +++ b/holmes/plugins/toolsets/investigator/investigator_instructions.jinja2 @@ -7,52 +7,50 @@ It is critical that you mark todos as completed as soon as you are done with a t Examples: -user: Run the build and fix any type errors -assistant: I'm going to use the TodoWrite tool to write the following items to the todo list: -- Run the build -- Fix any type errors - -I'm now going to run the build using Bash. - -Looks like I found 10 type errors. I'm going to use the TodoWrite tool to write 10 items to the todo list. +user: The invoice-reconciliation service is experiencing high latency and timeouts +assistant: I'll investigate the invoice-reconciliation service latency issue. Let me use the TodoWrite tool to plan this investigation: +- Check invoice-reconciliation service deployment status and pod health +- Examine recent logs for errors or performance issues +- Analyze resource utilization (CPU/memory) of invoice-reconciliation service pods +- Check service dependencies and external API response times +- Review recent configuration changes that might affect performance marking the first todo as in_progress -Let me start working on the first item... +Let me start by checking the deployment status... -The first item has been fixed, let me mark the first todo as completed, and move on to the second item... +I found the invoice-reconciliation service pods are experiencing CPU throttling. Let me mark this task as completed and investigate the resource limits... .. .. -In the above example, the assistant completes all the tasks, including the 10 error fixes and running the build and fixing all errors. +In the above example, the assistant systematically investigates the latency issue by checking infrastructure, logs, metrics, and configuration. -user: Help me write a new feature that allows users to track their usage metrics and export them to various formats +user: The shipment-tracker microservice pods keep restarting and we're seeing 502 errors -assistant: I'll help you implement a usage metrics tracking and export feature. Let me first use the TodoWrite tool to plan this task. -Adding the following todos to the todo list: -1. Research existing metrics tracking in the codebase -2. Design the metrics collection system -3. Implement core metrics tracking functionality -4. Create export functionality for different formats +assistant: I'll investigate the shipment-tracker service restart issue and 502 errors. Let me use the TodoWrite tool to plan this investigation: +1. Check shipment-tracker pod status and restart patterns +2. Examine pod logs around restart times to identify crash causes +3. Check service health checks and readiness/liveness probes +4. Analyze resource limits and requests for memory/CPU issues +5. Query distributed traces to analyze upstream/downstream service dependencies +6. Investigate any recent deployments or configuration changes -Let me start by researching the existing codebase to understand what metrics we might already be tracking and how we can build on that. +Let me start by examining the pod status and restart patterns. -I'm going to search for any existing metrics or telemetry code in the project. +I found the pods are crashing due to OOMKilled events. Let me mark this task as completed and investigate the memory usage patterns... -I've found some existing telemetry code. Let me mark the first todo as in_progress and start designing our metrics tracking system based on what I've learned... - -[Assistant continues implementing the feature step by step, marking todos as in_progress and completed as they go] +[Assistant continues investigating the issue step by step, marking todos as in_progress and completed as they go] # Doing tasks -The user will primarily request you perform reliability troubleshooting tasks. This includes finding bugs, finding misconfigurations, looking at observability data, including logs, traces, metrics, following investigation runbooks, searching for changes that might be related to incidents, adding new functionality, explaining code, and more. For these tasks the following steps are recommended: -- Use the TodoWrite tool to plan the task if required -- Use the available tools to understand the codebase, environment and the user's query. You are encouraged to use the tools extensively both in parallel and sequentially. -- Implement the solution using all tools available to you -- Verify the solution if possible. -- VERY IMPORTANT: When you have completed a task, validate that the answer you provided is correct. +The user will primarily request you perform reliability troubleshooting and incident investigation tasks. This includes analyzing observability data (logs, traces, metrics), identifying misconfigurations, finding root causes of outages, correlating incidents with recent changes, following investigation runbooks, and determining remediation steps. For these tasks the following steps are recommended: +- Use the TodoWrite tool to plan the investigation if required +- Use the available observability tools to gather data from multiple sources (metrics, logs, traces, configuration). You are encouraged to use the tools extensively both in parallel and sequentially. +- Analyze the collected data to identify patterns, correlations, and root causes +- Cross-reference findings across different data sources to validate conclusions +- VERY IMPORTANT: When you have completed an investigation, validate that your analysis is supported by the evidence collected. - Tool results and user messages may include tags. tags contain useful information and reminders. They are NOT part of the user's provided input or the tool result. @@ -81,7 +79,6 @@ IMPORTANT: Always use the TodoWrite tool to plan and track tasks throughout the - # TodoWrite Use this tool to create and manage a structured task list for your current coding session. This helps you track progress, organize complex tasks, and demonstrate thoroughness to the user. It also helps the user understand the progress of the task and overall progress of their requests. @@ -146,34 +143,34 @@ The assistant used the todo list because: -User: I need to implement these features for my e-commerce site: user registration, product catalog, shopping cart, and checkout flow. -Assistant: I'll help implement these features. First, let's add all the features to the todo list. -*Creates a todo list breaking down each feature into specific tasks based on the project architecture* -Assistant: Let's start with user registration. This will involve creating database models, API endpoints, and frontend forms that integrate with your existing authentication system. +User: We're seeing high error rates across multiple services after the latest deployment. Can you investigate what's causing the widespread issues? +Assistant: I'll investigate the widespread error rates after the deployment. Let me create a todo list to systematically track this multi-service investigation: +*Creates a todo list breaking down the investigation across different observability sources* +Assistant: Let's start by checking the deployment timeline and correlating it with error rate spikes using metrics and traces. The assistant used the todo list because: -1. The user provided multiple complex features to implement in a comma separated list -2. The todo list helps organize these large features into manageable tasks -3. This approach allows for tracking progress across the entire implementation +1. The user reported multiple services affected, requiring systematic investigation +2. The todo list helps organize investigation across different services and data sources +3. This approach ensures comprehensive analysis of metrics, logs, traces, and recent changes -User: Can you help optimize my React application? It's rendering slowly and has performance issues. -Assistant: I'll help optimize your React application. First, let me examine your codebase to identify potential performance bottlenecks. -*Reviews component structure, render patterns, state management, and data fetching* -Assistant: After analyzing your codebase, I've identified several performance issues. Let me create a todo list to track our optimization efforts. -*Creates todo list with items like: 1) Implement memoization for expensive calculations in ProductList, 2) Add virtualization for long lists in Dashboard, 3) Optimize image loading in Gallery component, 4) Fix state update loops in ShoppingCart, 5) Review bundle size and implement code splitting* -Let's start by implementing memoization for the expensive calculations in your ProductList component. +User: The inventory-validation service is showing high CPU usage and slow response times. Can you help troubleshoot this performance issue? +Assistant: I'll investigate the inventory-validation service performance issue. First, let me gather observability data to identify the root cause. +*Queries metrics to see CPU patterns, memory usage, and request latency trends* +Assistant: I found CPU spikes correlating with specific API endpoints. Let me create a todo list to systematically investigate this performance degradation. +*Creates todo list with items like: 1) Analyze CPU and memory metrics over time, 2) Check database query performance and connection pooling, 3) Examine distributed traces for slow operations, 4) Review recent code changes that might affect performance, 5) Check for resource contention with other services* +Let's start by analyzing the detailed CPU metrics and correlating with request patterns. The assistant used the todo list because: -1. First, the assistant examined the codebase to identify specific performance issues -2. Based on this analysis, it identified multiple optimization opportunities -3. Performance optimization is a non-trivial task requiring multiple steps -4. The todo list helps methodically track improvements across different components -5. This systematic approach ensures all performance bottlenecks are addressed +1. First, the assistant gathered initial observability data to understand the scope +2. Based on this analysis, it identified multiple potential performance bottlenecks +3. Performance troubleshooting requires systematic investigation across multiple data sources +4. The todo list helps methodically track investigation across metrics, traces, logs, and configuration +5. This systematic approach ensures comprehensive analysis of all potential causes @@ -239,16 +236,13 @@ The assistant did not use the todo list because this is a single command executi - Update task status in real-time as you work - Mark tasks complete IMMEDIATELY after finishing (don't batch completions) - Only have ONE task in_progress at any time - - Complete current tasks before starting new ones - - Remove tasks that are no longer relevant from the list entirely 3. **Task Completion Requirements**: - ONLY mark a task as completed when you have FULLY accomplished it - If you encounter errors, blockers, or cannot finish, keep the task as in_progress - When blocked, create a new task describing what needs to be resolved - Never mark a task as completed if: - - Tests are failing - - Implementation is partial + - Investigation is partial - You encountered unresolved errors - You couldn't find necessary files or dependencies From 0301614e0e3ff085654f4fd1dfb7a26c126394f7 Mon Sep 17 00:00:00 2001 From: Arik Alon Date: Mon, 18 Aug 2025 09:33:46 +0300 Subject: [PATCH 14/17] fix tests --- holmes/core/prompt.py | 18 +++++++++++------- tests/core/test_prompt.py | 13 ++++++++++--- tests/core/test_todo_manager.py | 5 ----- tests/core/test_todo_write_tool.py | 6 +++--- 4 files changed, 24 insertions(+), 18 deletions(-) diff --git a/holmes/core/prompt.py b/holmes/core/prompt.py index 477a6d0a70..af7ac61c4b 100644 --- a/holmes/core/prompt.py +++ b/holmes/core/prompt.py @@ -25,6 +25,16 @@ def append_all_files_to_user_prompt( return user_prompt +def get_tasks_management_system_reminder() -> str: + return ( + "\n\n\nIMPORTANT: You have access to the TodoWrite tool. It creates a TodoList, in order to track progress. It's very important. You MUST use it:\n1. FIRST: Ask your self which sub problems you need to solve in order to answer the question." + "Do this, BEFORE any other tools\n2. " + "AFTER EVERY TOOL CALL: If required, update the TodoList\n3. " + "\n\nFAILURE TO UPDATE TodoList = INCOMPLETE INVESTIGATION\n\n" + "Example flow:\n- Think and divide to sub problems → create TodoList → Perform each task on the list → Update list → Verify your solution\n" + ) + + def build_initial_ask_messages( console: Console, initial_user_prompt: str, @@ -61,13 +71,7 @@ def build_initial_ask_messages( console, initial_user_prompt, file_paths ) - user_prompt_with_files += ( - "\n\n\nIMPORTANT: You have access to the TodoWrite tool. It creates a TodoList, in order to track progress. It's very important. You MUST use it:\n1. FIRST: Ask your self which sub problems you need to solve in order to answer the question." - "Do this, BEFORE any other tools\n2. " - "AFTER EVERY TOOL CALL: If required, update the TodoList\n3. " - "\n\nFAILURE TO UPDATE TodoList = INCOMPLETE INVESTIGATION\n\n" - "Example flow:\n- Think and divide to sub problems → create TodoList → Perform each task on the list → Update list → Verify your solution\n" - ) + user_prompt_with_files += get_tasks_management_system_reminder() messages = [ {"role": "system", "content": system_prompt_rendered}, {"role": "user", "content": user_prompt_with_files}, diff --git a/tests/core/test_prompt.py b/tests/core/test_prompt.py index cc0892a5e4..7bcc54c602 100644 --- a/tests/core/test_prompt.py +++ b/tests/core/test_prompt.py @@ -8,6 +8,7 @@ build_initial_ask_messages, append_file_to_user_prompt, append_all_files_to_user_prompt, + get_tasks_management_system_reminder, ) @@ -38,7 +39,9 @@ def test_build_initial_ask_messages_basic(console, mock_tool_executor): assert len(messages) == 2 assert messages[0]["role"] == "system" assert messages[1]["role"] == "user" - assert messages[1]["content"] == "Test prompt" + assert ( + messages[1]["content"] == "Test prompt" + get_tasks_management_system_reminder() + ) def test_build_initial_ask_messages_with_system_prompt_additions( @@ -61,7 +64,9 @@ def test_build_initial_ask_messages_with_system_prompt_additions( # Check for unique word from the system additions assert "Additional" in messages[0]["content"] assert messages[1]["role"] == "user" - assert messages[1]["content"] == "Test prompt" + assert ( + messages[1]["content"] == "Test prompt" + get_tasks_management_system_reminder() + ) def test_build_initial_ask_messages_with_file(console, mock_tool_executor, tmp_path): @@ -109,7 +114,9 @@ def test_build_initial_ask_messages_with_runbooks(console, mock_tool_executor): assert messages[0]["role"] == "system" # The runbook should be passed to the template context assert messages[1]["role"] == "user" - assert messages[1]["content"] == "Test prompt" + assert ( + messages[1]["content"] == "Test prompt" + get_tasks_management_system_reminder() + ) def test_build_initial_ask_messages_all_parameters( diff --git a/tests/core/test_todo_manager.py b/tests/core/test_todo_manager.py index e0b33dc7d5..091b58ba0c 100644 --- a/tests/core/test_todo_manager.py +++ b/tests/core/test_todo_manager.py @@ -73,11 +73,6 @@ def test_prompt_formatting(self): assert "[~]" in prompt_context # in_progress assert "[✓]" in prompt_context # completed - # Check priority - assert "(HIGH)" in prompt_context - assert "(MED)" in prompt_context - assert "(LOW)" in prompt_context - def test_session_clearing(self): """Test clearing session tasks.""" manager = TodoListManager() diff --git a/tests/core/test_todo_write_tool.py b/tests/core/test_todo_write_tool.py index 940b320f12..79b754eff5 100644 --- a/tests/core/test_todo_write_tool.py +++ b/tests/core/test_todo_write_tool.py @@ -90,11 +90,11 @@ def test_get_parameterized_one_liner(self): params = {"todos": [{"content": "task1"}, {"content": "task2"}]} one_liner = tool.get_parameterized_one_liner(params) - assert "2 investigation tasks" in one_liner + assert f"{params.get('todos')} investigation tasks" in one_liner params = {"todos": []} one_liner = tool.get_parameterized_one_liner(params) - assert "0 investigation tasks" in one_liner + assert f"{params.get('todos')} investigation tasks" in one_liner def test_task_status_enum(self): """Test TaskStatus enum values.""" @@ -105,7 +105,7 @@ def test_task_status_enum(self): def test_openai_format(self): """Test that the tool generates correct OpenAI format.""" tool = TodoWriteTool() - openai_format = tool.get_openai_format() + openai_format = tool.get_openai_format("azure/gpt-4o") assert openai_format["type"] == "function" assert openai_format["function"]["name"] == "TodoWrite" From 633e1855a2b2758cbc96b891ff98c9662ed625b6 Mon Sep 17 00:00:00 2001 From: Arik Alon Date: Mon, 18 Aug 2025 10:22:11 +0300 Subject: [PATCH 15/17] cr fix --- holmes/core/openai_formatting.py | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/holmes/core/openai_formatting.py b/holmes/core/openai_formatting.py index 0aee7a5487..c4a235d130 100644 --- a/holmes/core/openai_formatting.py +++ b/holmes/core/openai_formatting.py @@ -53,10 +53,9 @@ def type_to_open_ai_schema(param_attributes: Any, strict_mode: bool) -> dict[str if match.group("inner_type"): inner_type = match.group("inner_type") if inner_type == "object": - items_obj: dict[str, Any] = {"type": "object"} - if strict_mode: - items_obj["additionalProperties"] = False - type_obj = {"type": "array", "items": items_obj} + raise ValueError( + "object inner type must have schema. Use ToolParameter.items" + ) else: type_obj = {"type": "array", "items": {"type": inner_type}} else: From ba127fe2eb2e88843b801ccdf00f90cf0488a71f Mon Sep 17 00:00:00 2001 From: Arik Alon Date: Mon, 18 Aug 2025 11:13:24 +0300 Subject: [PATCH 16/17] cr fix --- holmes/core/todo_manager.py | 2 +- .../investigator/core_investigation.py | 19 +++++++++++-------- .../investigator_instructions.jinja2 | 14 -------------- .../test_case.yaml | 9 --------- .../toolsets.yaml | 3 --- 5 files changed, 12 insertions(+), 35 deletions(-) delete mode 100644 tests/llm/fixtures/test_ask_holmes/150_what_is_wrong_with_mimir/test_case.yaml delete mode 100644 tests/llm/fixtures/test_ask_holmes/150_what_is_wrong_with_mimir/toolsets.yaml diff --git a/holmes/core/todo_manager.py b/holmes/core/todo_manager.py index 1aaf4aeb2f..5f7a740374 100644 --- a/holmes/core/todo_manager.py +++ b/holmes/core/todo_manager.py @@ -12,7 +12,7 @@ class TodoListManager: def __init__(self): self._sessions: Dict[str, List[Task]] = {} - self._lock = Lock() + self._lock: Lock = Lock() def get_session_tasks(self, session_id: str) -> List[Task]: with self._lock: diff --git a/holmes/plugins/toolsets/investigator/core_investigation.py b/holmes/plugins/toolsets/investigator/core_investigation.py index f6890d2457..b72e93df05 100644 --- a/holmes/plugins/toolsets/investigator/core_investigation.py +++ b/holmes/plugins/toolsets/investigator/core_investigation.py @@ -49,20 +49,23 @@ def print_tasks_table(self, tasks): logging.info("No tasks in the investigation plan.") return - max_id_width = max(len(str(task.id)) for task in tasks) - max_content_width = max(len(task.content) for task in tasks) - max_status_width = max(len(task.status.value) for task in tasks) - - id_width = max(max_id_width, 2) - content_width = max(max_content_width, 7) - status_width = max(max_status_width, 6) - status_icons = { "pending": "[ ]", "in_progress": "[~]", "completed": "[✓]", } + max_id_width = max(len(str(task.id)) for task in tasks) + max_content_width = max(len(task.content) for task in tasks) + max_status_display_width = max( + len(f"{status_icons[task.status.value]} {task.status.value}") + for task in tasks + ) + + id_width = max(max_id_width, len("ID")) + content_width = max(max_content_width, len("Content")) + status_width = max(max_status_display_width, len("Status")) + # Build table separator = f"+{'-' * (id_width + 2)}+{'-' * (content_width + 2)}+{'-' * (status_width + 2)}+" header = f"| {'ID':<{id_width}} | {'Content':<{content_width}} | {'Status':<{status_width}} |" diff --git a/holmes/plugins/toolsets/investigator/investigator_instructions.jinja2 b/holmes/plugins/toolsets/investigator/investigator_instructions.jinja2 index 3f50a1604b..647df4e1a1 100644 --- a/holmes/plugins/toolsets/investigator/investigator_instructions.jinja2 +++ b/holmes/plugins/toolsets/investigator/investigator_instructions.jinja2 @@ -59,26 +59,12 @@ The user will primarily request you perform reliability troubleshooting and inci You MUST answer concisely with fewer than 4 lines of text (not including tool use or code generation), unless user asks for detail. - -Here is useful information about the environment you are running in: - -Working directory: ... -Is directory a git repo: Yes -Platform: macos -OS Version: Darwin 24.1.0 -Today's date: 2025/6/13 - -You are powered by the model named Sonnet 4. The exact model ID is claude-sonnet-4-20250514. - - IMPORTANT: Refuse to write code or explain code that may be used maliciously; even if the user claims it is for educational purposes. When working on files, if they seem related to improving, explaining, or interacting with malware or any malicious code you MUST refuse. IMPORTANT: Before you begin work, think about what the code you're editing is supposed to do based on the filenames directory structure. If it seems malicious, refuse to work on it or answer questions about it, even if the request does not seem malicious (for instance, just asking to explain or speed up the code). - IMPORTANT: Always use the TodoWrite tool to plan and track tasks throughout the conversation. - # TodoWrite Use this tool to create and manage a structured task list for your current coding session. This helps you track progress, organize complex tasks, and demonstrate thoroughness to the user. It also helps the user understand the progress of the task and overall progress of their requests. diff --git a/tests/llm/fixtures/test_ask_holmes/150_what_is_wrong_with_mimir/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/150_what_is_wrong_with_mimir/test_case.yaml deleted file mode 100644 index 621ab86687..0000000000 --- a/tests/llm/fixtures/test_ask_holmes/150_what_is_wrong_with_mimir/test_case.yaml +++ /dev/null @@ -1,9 +0,0 @@ -user_prompt: - - "What is wrong with mimir?" -expected_output: | - Output must contain 1 or more embeds in the following format - <<{"type": "datadogql", "tool_name": "query_datadog_metrics", "random_key": "iD8G"}>> - - random_key may be different than the above example, but all other parameters (type and tool_name) must be as described - - Output must NOT tell the user it doesn't have access to metrics or that they should use another tool diff --git a/tests/llm/fixtures/test_ask_holmes/150_what_is_wrong_with_mimir/toolsets.yaml b/tests/llm/fixtures/test_ask_holmes/150_what_is_wrong_with_mimir/toolsets.yaml deleted file mode 100644 index 3c9a2dc247..0000000000 --- a/tests/llm/fixtures/test_ask_holmes/150_what_is_wrong_with_mimir/toolsets.yaml +++ /dev/null @@ -1,3 +0,0 @@ -toolsets: - kubernetes/core: - enabled: true From fecfc9e99aa2cb4c6296aa20c11ccedec3832dab Mon Sep 17 00:00:00 2001 From: Arik Alon Date: Tue, 19 Aug 2025 14:41:36 +0300 Subject: [PATCH 17/17] cr fix --- holmes/core/todo_manager.py | 16 ++++++++++------ .../toolsets/investigator/core_investigation.py | 2 -- 2 files changed, 10 insertions(+), 8 deletions(-) diff --git a/holmes/core/todo_manager.py b/holmes/core/todo_manager.py index 5f7a740374..b058f85961 100644 --- a/holmes/core/todo_manager.py +++ b/holmes/core/todo_manager.py @@ -41,11 +41,15 @@ def format_tasks_for_prompt(self, session_id: str) -> str: if not tasks: return "" - status_order = {"pending": 0, "in_progress": 1, "completed": 2} + status_order = { + TaskStatus.PENDING: 0, + TaskStatus.IN_PROGRESS: 1, + TaskStatus.COMPLETED: 2, + } sorted_tasks = sorted( tasks, - key=lambda t: (status_order.get(t.status.value, 3),), + key=lambda t: (status_order.get(t.status, 3),), ) lines = ["# CURRENT INVESTIGATION TASKS"] @@ -62,10 +66,10 @@ def format_tasks_for_prompt(self, session_id: str) -> str: for task in sorted_tasks: status_indicator = { - "pending": "[ ]", - "in_progress": "[~]", - "completed": "[✓]", - }.get(task.status.value, "[?]") + TaskStatus.PENDING: "[ ]", + TaskStatus.IN_PROGRESS: "[~]", + TaskStatus.COMPLETED: "[✓]", + }.get(task.status, "[?]") lines.append(f"{status_indicator} [{task.id}] {task.content}") diff --git a/holmes/plugins/toolsets/investigator/core_investigation.py b/holmes/plugins/toolsets/investigator/core_investigation.py index b72e93df05..2193a1a162 100644 --- a/holmes/plugins/toolsets/investigator/core_investigation.py +++ b/holmes/plugins/toolsets/investigator/core_investigation.py @@ -10,7 +10,6 @@ from holmes.core.tools import ( Toolset, ToolsetTag, - ToolsetStatusEnum, ToolParameter, Tool, StructuredToolResult, @@ -146,7 +145,6 @@ def __init__(self): tags=[ToolsetTag.CORE], is_default=True, ) - self.status = ToolsetStatusEnum.ENABLED logging.info("Core investigation toolset loaded") def get_example_config(self) -> Dict[str, Any]: