diff --git a/docs/ai-providers/anthropic.md b/docs/ai-providers/anthropic.md index 8f6f6c9529..1e15f54f89 100644 --- a/docs/ai-providers/anthropic.md +++ b/docs/ai-providers/anthropic.md @@ -21,6 +21,27 @@ You can also pass the API key directly as a command-line parameter: holmes ask "what pods are failing?" --model="anthropic/" --api-key="your-api-key" ``` +## Prompt Caching + +HolmesGPT adds Anthropic's prompt caching feature, which can significantly reduce costs and latency for repeated API calls with similar prompts. + +HolmesGPT automatically adds cache control to the last message in each API call. This caches everything from the beginning of the conversation up to that point, making subsequent calls with the same prefix much faster and cheaper. + +### How It Works + +- Anthropic uses prefix-based caching - it caches the exact sequence of messages up to the cache control point +- The cache has a 5-minute lifetime by default +- Cached content must be at least 1024 tokens to be effective +- You're charged for cache writes on the first call, but subsequent cache hits are much cheaper + +### Benefits in HolmesGPT + +Prompt caching is particularly effective for HolmesGPT because: + +- System prompts with tool definitions are large and static - perfect for caching +- Tool investigation loops reuse the same context multiple times +- Multi-step investigations benefit from cached conversation history + ## Additional Resources HolmesGPT uses the LiteLLM API to support Anthropic provider. Refer to [LiteLLM Anthropic docs](https://litellm.vercel.app/docs/providers/anthropic){:target="_blank"} for more details. diff --git a/docs/installation/python-installation.md b/docs/installation/python-installation.md index 1a154996e5..2778856592 100644 --- a/docs/installation/python-installation.md +++ b/docs/installation/python-installation.md @@ -48,7 +48,6 @@ messages = build_initial_ask_messages( initial_user_prompt=question, file_paths=None, tool_executor=ai.tool_executor, - investigation_id=ai.investigation_id, runbooks=config.get_runbook_catalog(), system_prompt_additions=None ) @@ -130,7 +129,6 @@ def main(): initial_user_prompt=question, file_paths=None, tool_executor=ai.tool_executor, - investigation_id=ai.investigation_id, runbooks=config.get_runbook_catalog(), system_prompt_additions=None ) @@ -224,7 +222,6 @@ def main(): initial_user_prompt=first_question, file_paths=None, tool_executor=ai.tool_executor, - investigation_id=ai.investigation_id, runbooks=config.get_runbook_catalog(), system_prompt_additions=None ) diff --git a/holmes/common/env_vars.py b/holmes/common/env_vars.py index a476a9e2cc..69054c71b4 100644 --- a/holmes/common/env_vars.py +++ b/holmes/common/env_vars.py @@ -67,3 +67,5 @@ def load_bool(env_var, default: Optional[bool]) -> Optional[bool]: # When using the bash tool, setting BASH_TOOL_UNSAFE_ALLOW_ALL will skip any command validation and run any command requested by the LLM BASH_TOOL_UNSAFE_ALLOW_ALL = load_bool("BASH_TOOL_UNSAFE_ALLOW_ALL", False) + +LOG_LLM_USAGE_RESPONSE = load_bool("LOG_LLM_USAGE_RESPONSE", False) diff --git a/holmes/core/conversations.py b/holmes/core/conversations.py index 9a306f2160..c61cc612a6 100644 --- a/holmes/core/conversations.py +++ b/holmes/core/conversations.py @@ -133,7 +133,6 @@ def build_issue_chat_messages( "issue": issue_chat_request.issue_type, "toolsets": ai.tool_executor.toolsets, "cluster_name": config.cluster_name, - "investigation_id": ai.investigation_id, }, ) messages = [ @@ -154,7 +153,6 @@ def build_issue_chat_messages( "issue": issue_chat_request.issue_type, "toolsets": ai.tool_executor.toolsets, "cluster_name": config.cluster_name, - "investigation_id": ai.investigation_id, } system_prompt_without_tools = load_and_render_prompt( template_path, template_context_without_tools @@ -188,7 +186,6 @@ def build_issue_chat_messages( "issue": issue_chat_request.issue_type, "toolsets": ai.tool_executor.toolsets, "cluster_name": config.cluster_name, - "investigation_id": ai.investigation_id, } system_prompt_with_truncated_tools = load_and_render_prompt( template_path, truncated_template_context @@ -230,7 +227,6 @@ def build_issue_chat_messages( "issue": issue_chat_request.issue_type, "toolsets": ai.tool_executor.toolsets, "cluster_name": config.cluster_name, - "investigation_id": ai.investigation_id, } system_prompt_without_tools = load_and_render_prompt( template_path, template_context_without_tools @@ -254,7 +250,6 @@ def build_issue_chat_messages( "issue": issue_chat_request.issue_type, "toolsets": ai.tool_executor.toolsets, "cluster_name": config.cluster_name, - "investigation_id": ai.investigation_id, } system_prompt_with_truncated_tools = load_and_render_prompt( template_path, template_context @@ -279,7 +274,6 @@ def add_or_update_system_prompt( context = { "toolsets": ai.tool_executor.toolsets, "cluster_name": config.cluster_name, - "investigation_id": ai.investigation_id, } system_prompt = load_and_render_prompt(template_path, context) @@ -471,7 +465,6 @@ def build_workload_health_chat_messages( "resource": resource, "toolsets": ai.tool_executor.toolsets, "cluster_name": config.cluster_name, - "investigation_id": ai.investigation_id, }, ) messages = [ @@ -492,7 +485,6 @@ def build_workload_health_chat_messages( "resource": resource, "toolsets": ai.tool_executor.toolsets, "cluster_name": config.cluster_name, - "investigation_id": ai.investigation_id, } system_prompt_without_tools = load_and_render_prompt( template_path, template_context_without_tools @@ -526,7 +518,6 @@ def build_workload_health_chat_messages( "resource": resource, "toolsets": ai.tool_executor.toolsets, "cluster_name": config.cluster_name, - "investigation_id": ai.investigation_id, } system_prompt_with_truncated_tools = load_and_render_prompt( template_path, truncated_template_context @@ -568,7 +559,6 @@ def build_workload_health_chat_messages( "resource": resource, "toolsets": ai.tool_executor.toolsets, "cluster_name": config.cluster_name, - "investigation_id": ai.investigation_id, } system_prompt_without_tools = load_and_render_prompt( template_path, template_context_without_tools @@ -592,7 +582,6 @@ def build_workload_health_chat_messages( "resource": resource, "toolsets": ai.tool_executor.toolsets, "cluster_name": config.cluster_name, - "investigation_id": ai.investigation_id, } system_prompt_with_truncated_tools = load_and_render_prompt( template_path, template_context diff --git a/holmes/core/investigation.py b/holmes/core/investigation.py index 1440b63891..e81d440eb1 100644 --- a/holmes/core/investigation.py +++ b/holmes/core/investigation.py @@ -9,7 +9,6 @@ from holmes.core.supabase_dal import SupabaseDal from holmes.core.tracing import DummySpan, SpanType from holmes.utils.global_instructions import add_global_instructions_to_user_prompt -from holmes.core.todo_manager import get_todo_manager from holmes.core.investigation_structured_output import ( DEFAULT_SECTIONS, @@ -133,9 +132,6 @@ def get_investigation_context( else: logging.info("Structured output is disabled for this request") - todo_manager = get_todo_manager() - todo_context = todo_manager.format_tasks_for_prompt(ai.investigation_id) - system_prompt = load_and_render_prompt( investigate_request.prompt_template, { @@ -144,8 +140,6 @@ def get_investigation_context( "structured_output": request_structured_output_from_llm, "toolsets": ai.tool_executor.toolsets, "cluster_name": config.cluster_name, - "todo_list": todo_context, - "investigation_id": ai.investigation_id, }, ) diff --git a/holmes/core/llm.py b/holmes/core/llm.py index a25fc2fd00..516fea3967 100644 --- a/holmes/core/llm.py +++ b/holmes/core/llm.py @@ -229,9 +229,11 @@ def completion( ] # can be removed after next litelm version self.args.setdefault("temperature", temperature) + + self._add_cache_control_to_last_message(messages) + # Get the litellm module to use (wrapped or unwrapped) litellm_to_use = self.tracer.wrap_llm(litellm) if self.tracer else litellm - result = litellm_to_use.completion( model=self.model, api_key=self.api_key, @@ -266,3 +268,60 @@ def get_maximum_output_token(self) -> int: f"Couldn't find model's name {model_name} in litellm's model list, fallback to 4096 tokens for max_output_tokens" ) return 4096 + + def _add_cache_control_to_last_message( + self, messages: List[Dict[str, Any]] + ) -> None: + """ + Add cache_control to the last non-user message for Anthropic prompt caching. + Removes any existing cache_control from previous messages to avoid accumulation. + """ + # First, remove any existing cache_control from all messages + for msg in messages: + content = msg.get("content") + if isinstance(content, list): + for block in content: + if isinstance(block, dict) and "cache_control" in block: + del block["cache_control"] + logging.debug( + f"Removed existing cache_control from {msg.get('role')} message" + ) + + # Find the last non-user message to add cache_control to. + # Adding cache_control to user message requires changing its structure, so we avoid it + # This avoids breaking parse_messages_tags which only processes user messages + target_msg = None + for msg in reversed(messages): + if msg.get("role") != "user": + target_msg = msg + break + + if not target_msg: + logging.debug("No non-user message found for cache_control") + return + + content = target_msg.get("content") + + if content is None: + return + + if isinstance(content, str): + # Convert string to structured format with cache_control + target_msg["content"] = [ + { + "type": "text", + "text": content, + "cache_control": {"type": "ephemeral"}, + } + ] + logging.debug( + f"Added cache_control to {target_msg.get('role')} message (converted from string)" + ) + elif isinstance(content, list) and content: + # Add cache_control to the last content block + last_block = content[-1] + if isinstance(last_block, dict) and "type" in last_block: + last_block["cache_control"] = {"type": "ephemeral"} + logging.debug( + f"Added cache_control to {target_msg.get('role')} message (structured content)" + ) diff --git a/holmes/core/prompt.py b/holmes/core/prompt.py index af7ac61c4b..a0e5f8160f 100644 --- a/holmes/core/prompt.py +++ b/holmes/core/prompt.py @@ -40,7 +40,6 @@ def build_initial_ask_messages( initial_user_prompt: str, file_paths: Optional[List[Path]], tool_executor: Any, # ToolExecutor type - investigation_id: str, runbooks: Union[RunbookCatalog, Dict, None] = None, system_prompt_additions: Optional[str] = None, ) -> List[Dict]: @@ -60,7 +59,6 @@ def build_initial_ask_messages( "toolsets": tool_executor.toolsets, "runbooks": runbooks or {}, "system_prompt_additions": system_prompt_additions or "", - "investigation_id": investigation_id, } system_prompt_rendered = load_and_render_prompt( system_prompt_template, template_context diff --git a/holmes/core/todo_manager.py b/holmes/core/todo_manager.py deleted file mode 100644 index b058f85961..0000000000 --- a/holmes/core/todo_manager.py +++ /dev/null @@ -1,88 +0,0 @@ -from typing import Dict, List -from threading import Lock - -from holmes.plugins.toolsets.investigator.model import Task, TaskStatus - - -class TodoListManager: - """ - Session-based storage manager for investigation TodoLists. - Stores TodoLists per session and provides methods to get/update tasks. - """ - - def __init__(self): - self._sessions: Dict[str, List[Task]] = {} - self._lock: Lock = Lock() - - def get_session_tasks(self, session_id: str) -> List[Task]: - with self._lock: - return self._sessions.get(session_id, []).copy() - - def update_session_tasks(self, session_id: str, tasks: List[Task]) -> None: - with self._lock: - self._sessions[session_id] = tasks.copy() - - def clear_session(self, session_id: str) -> None: - with self._lock: - if session_id in self._sessions: - del self._sessions[session_id] - - def get_session_count(self) -> int: - with self._lock: - return len(self._sessions) - - def format_tasks_for_prompt(self, session_id: str) -> str: - """ - Format tasks for injection into system prompt. - Returns empty string if no tasks exist. - """ - tasks = self.get_session_tasks(session_id) - - if not tasks: - return "" - - status_order = { - TaskStatus.PENDING: 0, - TaskStatus.IN_PROGRESS: 1, - TaskStatus.COMPLETED: 2, - } - - sorted_tasks = sorted( - tasks, - key=lambda t: (status_order.get(t.status, 3),), - ) - - lines = ["# CURRENT INVESTIGATION TASKS"] - lines.append("") - - pending_count = sum(1 for t in tasks if t.status == TaskStatus.PENDING) - progress_count = sum(1 for t in tasks if t.status == TaskStatus.IN_PROGRESS) - completed_count = sum(1 for t in tasks if t.status == TaskStatus.COMPLETED) - - lines.append( - f"**Task Status**: {completed_count} completed, {progress_count} in progress, {pending_count} pending" - ) - lines.append("") - - for task in sorted_tasks: - status_indicator = { - TaskStatus.PENDING: "[ ]", - TaskStatus.IN_PROGRESS: "[~]", - TaskStatus.COMPLETED: "[✓]", - }.get(task.status, "[?]") - - lines.append(f"{status_indicator} [{task.id}] {task.content}") - - lines.append("") - lines.append( - "**Instructions**: Use TodoWrite tool to update task status as you work. Mark tasks as 'in_progress' when starting, 'completed' when finished." - ) - - return "\n".join(lines) - - -_todo_manager = TodoListManager() - - -def get_todo_manager() -> TodoListManager: - return _todo_manager diff --git a/holmes/core/todo_tasks_formatter.py b/holmes/core/todo_tasks_formatter.py new file mode 100644 index 0000000000..fbf39a4e59 --- /dev/null +++ b/holmes/core/todo_tasks_formatter.py @@ -0,0 +1,51 @@ +from typing import List + +from holmes.plugins.toolsets.investigator.model import Task, TaskStatus + + +def format_tasks(tasks: List[Task]) -> str: + """ + Format tasks for tool response + Returns empty string if no tasks exist. + """ + if not tasks: + return "" + + status_order = { + TaskStatus.PENDING: 0, + TaskStatus.IN_PROGRESS: 1, + TaskStatus.COMPLETED: 2, + } + + sorted_tasks = sorted( + tasks, + key=lambda t: (status_order.get(t.status, 3),), + ) + + lines = ["# CURRENT INVESTIGATION TASKS"] + lines.append("") + + pending_count = sum(1 for t in tasks if t.status == TaskStatus.PENDING) + progress_count = sum(1 for t in tasks if t.status == TaskStatus.IN_PROGRESS) + completed_count = sum(1 for t in tasks if t.status == TaskStatus.COMPLETED) + + lines.append( + f"**Task Status**: {completed_count} completed, {progress_count} in progress, {pending_count} pending" + ) + lines.append("") + + for task in sorted_tasks: + status_indicator = { + TaskStatus.PENDING: "[ ]", + TaskStatus.IN_PROGRESS: "[~]", + TaskStatus.COMPLETED: "[✓]", + }.get(task.status, "[?]") + + lines.append(f"{status_indicator} [{task.id}] {task.content}") + + lines.append("") + lines.append( + "**Instructions**: Use TodoWrite tool to update task status as you work. Mark tasks as 'in_progress' when starting, 'completed' when finished." + ) + + return "\n".join(lines) diff --git a/holmes/core/tool_calling_llm.py b/holmes/core/tool_calling_llm.py index 6d7fd2d2f4..2791c107b9 100644 --- a/holmes/core/tool_calling_llm.py +++ b/holmes/core/tool_calling_llm.py @@ -2,7 +2,6 @@ import json import logging import textwrap -import uuid from typing import Dict, List, Optional, Type, Union import sentry_sdk @@ -13,7 +12,11 @@ from pydantic import BaseModel, Field from rich.console import Console -from holmes.common.env_vars import TEMPERATURE, MAX_OUTPUT_TOKEN_RESERVATION +from holmes.common.env_vars import ( + TEMPERATURE, + MAX_OUTPUT_TOKEN_RESERVATION, + LOG_LLM_USAGE_RESPONSE, +) from holmes.core.investigation_structured_output import ( DEFAULT_SECTIONS, @@ -39,9 +42,6 @@ from holmes.core.tracing import DummySpan from holmes.utils.colors import AI_COLOR from holmes.utils.stream import StreamEvents, StreamMessage -from holmes.core.todo_manager import ( - get_todo_manager, -) # Create a named logger for cost tracking cost_logger = logging.getLogger("holmes.costs") @@ -94,6 +94,8 @@ def _process_cost_info( usage = getattr(full_response, "usage", {}) if usage: + if LOG_LLM_USAGE_RESPONSE: # shows stats on token cache usage + logging.info(f"LLM usage response:\n{usage}\n") prompt_toks = usage.get("prompt_tokens", 0) completion_toks = usage.get("completion_tokens", 0) total_toks = usage.get("total_tokens", 0) @@ -283,7 +285,6 @@ def __init__( self.max_steps = max_steps self.tracer = tracer self.llm = llm - self.investigation_id = str(uuid.uuid4()) def prompt_call( self, @@ -894,9 +895,6 @@ def investigate( "[bold]No runbooks found for this issue. Using default behaviour. (Add runbooks to guide the investigation.)[/bold]" ) - todo_manager = get_todo_manager() - todo_context = todo_manager.format_tasks_for_prompt(self.investigation_id) - system_prompt = load_and_render_prompt( prompt, { @@ -905,8 +903,6 @@ def investigate( "structured_output": request_structured_output_from_llm, "toolsets": self.tool_executor.toolsets, "cluster_name": self.cluster_name, - "todo_list": todo_context, - "investigation_id": self.investigation_id, }, ) diff --git a/holmes/interactive.py b/holmes/interactive.py index 4a98f7f014..0c6d6c5d38 100644 --- a/holmes/interactive.py +++ b/holmes/interactive.py @@ -1002,7 +1002,6 @@ def get_bottom_toolbar(): user_input, include_files, ai.tool_executor, - ai.investigation_id, runbooks, system_prompt_additions, ) diff --git a/holmes/main.py b/holmes/main.py index f1c07126c7..918829d975 100644 --- a/holmes/main.py +++ b/holmes/main.py @@ -308,7 +308,6 @@ def ask( prompt, # type: ignore include_file, ai.tool_executor, - ai.investigation_id, config.get_runbook_catalog(), system_prompt_additions, ) diff --git a/holmes/plugins/prompts/_general_instructions.jinja2 b/holmes/plugins/prompts/_general_instructions.jinja2 index d896743f36..17122a1935 100644 --- a/holmes/plugins/prompts/_general_instructions.jinja2 +++ b/holmes/plugins/prompts/_general_instructions.jinja2 @@ -62,10 +62,12 @@ * Follow ALL tasks in your plan - don't skip any tasks * Use task management to ensure you don't miss important investigation steps * If you discover additional steps during investigation, add them to your task list using TodoWrite +* When calling TodoWrite, you may ALSO call other tools in parallel to speed things up for your users and make them happy! +* On the first TodoWrite call, mark at least one task as in_progress, and start working on it in parallel. +* When calling TodoWrite for the first time, mark the tasks you started working on with ‘in_progress’ status. # Tool/function calls You are able to make tool calls / function calls. Recognise when a tool has already been called and reuse its result. If a tool call returns nothing, modify the parameters as required instead of repeating the tool call. When searching for resources in specific namespaces, test a cluster level tool to find the resource(s) and identify what namespace they are part of. -You are limited in use to a maximum of 5 tool calls for each specific tool. Therefore make sure are smart about what tools you call and how you call them. diff --git a/holmes/plugins/prompts/investigation_procedure.jinja2 b/holmes/plugins/prompts/investigation_procedure.jinja2 index e1271bccc6..d5d3129d3d 100644 --- a/holmes/plugins/prompts/investigation_procedure.jinja2 +++ b/holmes/plugins/prompts/investigation_procedure.jinja2 @@ -1,8 +1,3 @@ -{% if investigation_id %} -# Investigation ID for this session -Investigation id: {{ investigation_id }} -{% endif %} - CLARIFICATION REQUIREMENT: Before starting ANY investigation, if the user's question is ambiguous or lacks critical details, you MUST ask for clarification first. Do NOT create TodoWrite tasks for unclear questions. Only proceed with TodoWrite and investigation AFTER you have clear, specific requirements. @@ -16,8 +11,8 @@ MANDATORY Task Status Updates: - When completing a task: Call TodoWrite changing that task's status to "completed" PARALLEL EXECUTION RULES: -- When possible, work on multiple tasks at a time. If tasks depend on one another, do them one after the other. -- You MAY execute multiple INDEPENDENT tasks simultaneously +- When possible, work on multiple tasks at a time. Only when tasks depend on one another, do them one after the other. +- You SHOULD execute multiple INDEPENDENT tasks simultaneously - Mark multiple tasks as "in_progress" if they don't depend on each other - Wait for dependent tasks to complete before starting tasks that need their results - Always use a single TodoWrite call to update multiple task statuses @@ -84,17 +79,12 @@ If you see ANY `[ ] pending` or `[~] in_progress` tasks, DO NOT provide final an {"id": "3", "content": "Check resources", "status": "pending"} ]) - -{% if todo_list %} -{{ todo_list }} -{% endif %} - # MANDATORY Multi-Phase Investigation Process For ANY question requiring investigation, you MUST follow this structured approach: ## Phase 1: Initial Investigation -1. **IMMEDIATELY START with TodoWrite**: Create initial investigation task list +1. **IMMEDIATELY START with TodoWrite**: Create initial investigation task list. Already start working on tasks. Mark the tasks you're working on as in_progress. 2. **Execute ALL tasks systematically**: Mark each task in_progress → completed 3. **Complete EVERY task** in the current list before proceeding diff --git a/holmes/plugins/toolsets/aws.yaml b/holmes/plugins/toolsets/aws.yaml index c8ba2a72ec..8ee8ca240c 100644 --- a/holmes/plugins/toolsets/aws.yaml +++ b/holmes/plugins/toolsets/aws.yaml @@ -44,6 +44,10 @@ toolsets: description: "Read access to Amazon RDS resources" docs_url: "https://docs.robusta.dev/master/configuration/holmesgpt/toolsets/aws.html#rds" icon_url: "https://upload.wikimedia.org/wikipedia/commons/9/93/Amazon_Web_Services_Logo.svg" + llm_instructions: | + You have access to information on RDS resources. + Use this toolset to get extra information on RDS resources on AWS. + When investigating RDS resources ALWAYS fetch additional information if possible. tags: - cli prerequisites: @@ -56,8 +60,8 @@ toolsets: command: "aws rds describe-events" - name: "aws_rds_describe_instance" - description: "Get the configuration of a RDS instance" - user_description: "Get the configuration of a RDS instance" + description: "Get the configuration of a RDS instance, its status and availability stats. Runs 'aws rds describe-db-instances' for a specific instance." + user_description: "Get the configuration of an RDS instance" command: "aws rds describe-db-instances --db-instance-identifier '{{ db_instance_identifier }}'" - name: "aws_rds_describe_instances" @@ -66,7 +70,7 @@ toolsets: command: "aws rds describe-db-instances" - name: "aws_rds_describe_logs" - description: "Describe all available logs for an AWS RDS instance." + description: "Describe all available logs for an AWS RDS instance. Runs 'aws rds describe-db-log-files' for a specific instance." user_description: "list available RDS logs (e.g. slow query logs)" command: "aws rds describe-db-log-files --db-instance-identifier '{{ db_instance_identifier }}'" diff --git a/holmes/plugins/toolsets/investigator/core_investigation.py b/holmes/plugins/toolsets/investigator/core_investigation.py index 2193a1a162..1f831d6798 100644 --- a/holmes/plugins/toolsets/investigator/core_investigation.py +++ b/holmes/plugins/toolsets/investigator/core_investigation.py @@ -3,10 +3,7 @@ from typing import Any, Dict from uuid import uuid4 -from holmes.core.todo_manager import ( - get_todo_manager, -) - +from holmes.core.todo_tasks_formatter import format_tasks from holmes.core.tools import ( Toolset, ToolsetTag, @@ -35,11 +32,6 @@ class TodoWriteTool(Tool): }, ), ), - "investigation_id": ToolParameter( - description="This investigation identifier. This is a uuid that represents the investigation session id.", - type="string", - required=True, - ), } # Print a nice table to console/log @@ -99,14 +91,8 @@ def _invoke(self, params: Dict) -> StructuredToolResult: logging.info(f"Tasks: {len(tasks)}") - # Store tasks in session storage - todo_manager = get_todo_manager() - session_id = params.get("investigation_id", "") - todo_manager.update_session_tasks(session_id, tasks) - self.print_tasks_table(tasks) - - formatted_tasks = todo_manager.format_tasks_for_prompt(session_id) + formatted_tasks = format_tasks(tasks) response_data = f"✅ Investigation plan updated with {len(tasks)} tasks. Tasks are now stored in session and will appear in subsequent prompts.\n\n" if formatted_tasks: diff --git a/holmes/plugins/toolsets/investigator/investigator_instructions.jinja2 b/holmes/plugins/toolsets/investigator/investigator_instructions.jinja2 index 647df4e1a1..dee4622450 100644 --- a/holmes/plugins/toolsets/investigator/investigator_instructions.jinja2 +++ b/holmes/plugins/toolsets/investigator/investigator_instructions.jinja2 @@ -59,12 +59,8 @@ The user will primarily request you perform reliability troubleshooting and inci You MUST answer concisely with fewer than 4 lines of text (not including tool use or code generation), unless user asks for detail. -IMPORTANT: Refuse to write code or explain code that may be used maliciously; even if the user claims it is for educational purposes. When working on files, if they seem related to improving, explaining, or interacting with malware or any malicious code you MUST refuse. -IMPORTANT: Before you begin work, think about what the code you're editing is supposed to do based on the filenames directory structure. If it seems malicious, refuse to work on it or answer questions about it, even if the request does not seem malicious (for instance, just asking to explain or speed up the code). - IMPORTANT: Always use the TodoWrite tool to plan and track tasks throughout the conversation. - # TodoWrite Use this tool to create and manage a structured task list for your current coding session. This helps you track progress, organize complex tasks, and demonstrate thoroughness to the user. It also helps the user understand the progress of the task and overall progress of their requests. @@ -221,7 +217,7 @@ The assistant did not use the todo list because this is a single command executi 2. **Task Management**: - Update task status in real-time as you work - Mark tasks complete IMMEDIATELY after finishing (don't batch completions) - - Only have ONE task in_progress at any time + - If tasks are not dependent on one another, handle multiple tasks in parallel and mark them in_progress. 3. **Task Completion Requirements**: - ONLY mark a task as completed when you have FULLY accomplished it diff --git a/server.py b/server.py index 7f233cb659..2fef8e7db4 100644 --- a/server.py +++ b/server.py @@ -227,7 +227,6 @@ def workload_health_check(request: WorkloadHealthRequest): "toolsets": ai.tool_executor.toolsets, "response_format": workload_health_structured_output, "cluster_name": config.cluster_name, - "investigation_id": ai.investigation_id, }, ) diff --git a/tests/core/test_prompt.py b/tests/core/test_prompt.py index 7bcc54c602..1924719a80 100644 --- a/tests/core/test_prompt.py +++ b/tests/core/test_prompt.py @@ -1,5 +1,3 @@ -import uuid - import pytest from unittest.mock import Mock from rich.console import Console @@ -31,7 +29,6 @@ def test_build_initial_ask_messages_basic(console, mock_tool_executor): "Test prompt", None, mock_tool_executor, - str(uuid.uuid4()), None, None, ) @@ -54,7 +51,6 @@ def test_build_initial_ask_messages_with_system_prompt_additions( "Test prompt", None, mock_tool_executor, - str(uuid.uuid4()), None, system_additions, ) @@ -80,7 +76,6 @@ def test_build_initial_ask_messages_with_file(console, mock_tool_executor, tmp_p "Test prompt", [test_file], mock_tool_executor, - str(uuid.uuid4()), None, None, ) @@ -105,7 +100,6 @@ def test_build_initial_ask_messages_with_runbooks(console, mock_tool_executor): "Test prompt", None, mock_tool_executor, - str(uuid.uuid4()), runbooks, None, ) @@ -135,7 +129,6 @@ def test_build_initial_ask_messages_all_parameters( "Test prompt", [test_file], mock_tool_executor, - str(uuid.uuid4()), runbooks, system_additions, ) diff --git a/tests/core/test_todo_manager.py b/tests/core/test_todo_manager.py deleted file mode 100644 index 091b58ba0c..0000000000 --- a/tests/core/test_todo_manager.py +++ /dev/null @@ -1,93 +0,0 @@ -from holmes.core.todo_manager import ( - TodoListManager, - get_todo_manager, -) -from holmes.plugins.toolsets.investigator.model import Task, TaskStatus - - -class TestTodoListManager: - def test_manager_creation(self): - """Test that TodoListManager can be created.""" - manager = TodoListManager() - assert manager.get_session_count() == 0 - - def test_empty_session(self): - """Test handling of empty session.""" - manager = TodoListManager() - tasks = manager.get_session_tasks("non-existent-session") - assert tasks == [] - - prompt_context = manager.format_tasks_for_prompt("non-existent-session") - assert prompt_context == "" - - def test_session_task_management(self): - """Test adding and retrieving tasks from session.""" - manager = TodoListManager() - session_id = "test-session" - - # Create test tasks - tasks = [ - Task(id="1", content="Task 1", status=TaskStatus.PENDING), - Task(id="2", content="Task 2", status=TaskStatus.IN_PROGRESS), - Task(id="3", content="Task 3", status=TaskStatus.COMPLETED), - ] - - # Update session tasks - manager.update_session_tasks(session_id, tasks) - assert manager.get_session_count() == 1 - - # Retrieve tasks - retrieved_tasks = manager.get_session_tasks(session_id) - assert len(retrieved_tasks) == 3 - assert retrieved_tasks[0].content == "Task 1" - assert retrieved_tasks[1].content == "Task 2" - assert retrieved_tasks[2].content == "Task 3" - - def test_prompt_formatting(self): - """Test that tasks are formatted correctly for prompt injection.""" - manager = TodoListManager() - session_id = "test-prompt-session" - - tasks = [ - Task(id="1", content="Check system", status=TaskStatus.PENDING), - Task(id="2", content="Review logs", status=TaskStatus.IN_PROGRESS), - Task(id="3", content="Write report", status=TaskStatus.COMPLETED), - ] - - manager.update_session_tasks(session_id, tasks) - prompt_context = manager.format_tasks_for_prompt(session_id) - - # Check structure - assert "# CURRENT INVESTIGATION TASKS" in prompt_context - assert ( - "**Task Status**: 1 completed, 1 in progress, 1 pending" in prompt_context - ) - - # Check tasks appear - assert "Check system" in prompt_context - assert "Review logs" in prompt_context - assert "Write report" in prompt_context - - # Check indicators - assert "[ ]" in prompt_context # pending - assert "[~]" in prompt_context # in_progress - assert "[✓]" in prompt_context # completed - - def test_session_clearing(self): - """Test clearing session tasks.""" - manager = TodoListManager() - session_id = "test-clear-session" - - tasks = [Task(id="1", content="Task 1", status=TaskStatus.PENDING)] - manager.update_session_tasks(session_id, tasks) - assert manager.get_session_count() == 1 - - manager.clear_session(session_id) - assert manager.get_session_count() == 0 - assert manager.get_session_tasks(session_id) == [] - - def test_global_manager_instance(self): - """Test that get_todo_manager returns the same instance.""" - manager1 = get_todo_manager() - manager2 = get_todo_manager() - assert manager1 is manager2 diff --git a/tests/llm/test_ask_holmes.py b/tests/llm/test_ask_holmes.py index 6f17d11bb3..3f4cea33b4 100644 --- a/tests/llm/test_ask_holmes.py +++ b/tests/llm/test_ask_holmes.py @@ -208,7 +208,6 @@ def ask_holmes( test_case.user_prompt, None, ai.tool_executor, - ai.investigation_id, runbooks, ) else: