Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
21 changes: 21 additions & 0 deletions docs/ai-providers/anthropic.md
Original file line number Diff line number Diff line change
Expand Up @@ -21,6 +21,27 @@ You can also pass the API key directly as a command-line parameter:
holmes ask "what pods are failing?" --model="anthropic/<your-claude-model>" --api-key="your-api-key"
```

## Prompt Caching

HolmesGPT adds Anthropic's prompt caching feature, which can significantly reduce costs and latency for repeated API calls with similar prompts.

HolmesGPT automatically adds cache control to the last message in each API call. This caches everything from the beginning of the conversation up to that point, making subsequent calls with the same prefix much faster and cheaper.

### How It Works

- Anthropic uses prefix-based caching - it caches the exact sequence of messages up to the cache control point
- The cache has a 5-minute lifetime by default
- Cached content must be at least 1024 tokens to be effective
- You're charged for cache writes on the first call, but subsequent cache hits are much cheaper

### Benefits in HolmesGPT

Prompt caching is particularly effective for HolmesGPT because:

- System prompts with tool definitions are large and static - perfect for caching
- Tool investigation loops reuse the same context multiple times
- Multi-step investigations benefit from cached conversation history

Comment thread
arikalon1 marked this conversation as resolved.
## Additional Resources

HolmesGPT uses the LiteLLM API to support Anthropic provider. Refer to [LiteLLM Anthropic docs](https://litellm.vercel.app/docs/providers/anthropic){:target="_blank"} for more details.
3 changes: 0 additions & 3 deletions docs/installation/python-installation.md
Original file line number Diff line number Diff line change
Expand Up @@ -48,7 +48,6 @@ messages = build_initial_ask_messages(
initial_user_prompt=question,
file_paths=None,
tool_executor=ai.tool_executor,
investigation_id=ai.investigation_id,
runbooks=config.get_runbook_catalog(),
system_prompt_additions=None
)
Expand Down Expand Up @@ -130,7 +129,6 @@ def main():
initial_user_prompt=question,
file_paths=None,
tool_executor=ai.tool_executor,
investigation_id=ai.investigation_id,
runbooks=config.get_runbook_catalog(),
system_prompt_additions=None
)
Expand Down Expand Up @@ -224,7 +222,6 @@ def main():
initial_user_prompt=first_question,
file_paths=None,
tool_executor=ai.tool_executor,
investigation_id=ai.investigation_id,
runbooks=config.get_runbook_catalog(),
system_prompt_additions=None
)
Expand Down
2 changes: 2 additions & 0 deletions holmes/common/env_vars.py
Original file line number Diff line number Diff line change
Expand Up @@ -67,3 +67,5 @@ def load_bool(env_var, default: Optional[bool]) -> Optional[bool]:

# When using the bash tool, setting BASH_TOOL_UNSAFE_ALLOW_ALL will skip any command validation and run any command requested by the LLM
BASH_TOOL_UNSAFE_ALLOW_ALL = load_bool("BASH_TOOL_UNSAFE_ALLOW_ALL", False)

LOG_LLM_USAGE_RESPONSE = load_bool("LOG_LLM_USAGE_RESPONSE", False)
11 changes: 0 additions & 11 deletions holmes/core/conversations.py
Original file line number Diff line number Diff line change
Expand Up @@ -133,7 +133,6 @@ def build_issue_chat_messages(
"issue": issue_chat_request.issue_type,
"toolsets": ai.tool_executor.toolsets,
"cluster_name": config.cluster_name,
"investigation_id": ai.investigation_id,
},
)
messages = [
Expand All @@ -154,7 +153,6 @@ def build_issue_chat_messages(
"issue": issue_chat_request.issue_type,
"toolsets": ai.tool_executor.toolsets,
"cluster_name": config.cluster_name,
"investigation_id": ai.investigation_id,
}
system_prompt_without_tools = load_and_render_prompt(
template_path, template_context_without_tools
Expand Down Expand Up @@ -188,7 +186,6 @@ def build_issue_chat_messages(
"issue": issue_chat_request.issue_type,
"toolsets": ai.tool_executor.toolsets,
"cluster_name": config.cluster_name,
"investigation_id": ai.investigation_id,
}
system_prompt_with_truncated_tools = load_and_render_prompt(
template_path, truncated_template_context
Expand Down Expand Up @@ -230,7 +227,6 @@ def build_issue_chat_messages(
"issue": issue_chat_request.issue_type,
"toolsets": ai.tool_executor.toolsets,
"cluster_name": config.cluster_name,
"investigation_id": ai.investigation_id,
}
system_prompt_without_tools = load_and_render_prompt(
template_path, template_context_without_tools
Expand All @@ -254,7 +250,6 @@ def build_issue_chat_messages(
"issue": issue_chat_request.issue_type,
"toolsets": ai.tool_executor.toolsets,
"cluster_name": config.cluster_name,
"investigation_id": ai.investigation_id,
}
system_prompt_with_truncated_tools = load_and_render_prompt(
template_path, template_context
Expand All @@ -279,7 +274,6 @@ def add_or_update_system_prompt(
context = {
"toolsets": ai.tool_executor.toolsets,
"cluster_name": config.cluster_name,
"investigation_id": ai.investigation_id,
}

system_prompt = load_and_render_prompt(template_path, context)
Expand Down Expand Up @@ -471,7 +465,6 @@ def build_workload_health_chat_messages(
"resource": resource,
"toolsets": ai.tool_executor.toolsets,
"cluster_name": config.cluster_name,
"investigation_id": ai.investigation_id,
},
)
messages = [
Expand All @@ -492,7 +485,6 @@ def build_workload_health_chat_messages(
"resource": resource,
"toolsets": ai.tool_executor.toolsets,
"cluster_name": config.cluster_name,
"investigation_id": ai.investigation_id,
}
system_prompt_without_tools = load_and_render_prompt(
template_path, template_context_without_tools
Expand Down Expand Up @@ -526,7 +518,6 @@ def build_workload_health_chat_messages(
"resource": resource,
"toolsets": ai.tool_executor.toolsets,
"cluster_name": config.cluster_name,
"investigation_id": ai.investigation_id,
}
system_prompt_with_truncated_tools = load_and_render_prompt(
template_path, truncated_template_context
Expand Down Expand Up @@ -568,7 +559,6 @@ def build_workload_health_chat_messages(
"resource": resource,
"toolsets": ai.tool_executor.toolsets,
"cluster_name": config.cluster_name,
"investigation_id": ai.investigation_id,
}
system_prompt_without_tools = load_and_render_prompt(
template_path, template_context_without_tools
Expand All @@ -592,7 +582,6 @@ def build_workload_health_chat_messages(
"resource": resource,
"toolsets": ai.tool_executor.toolsets,
"cluster_name": config.cluster_name,
"investigation_id": ai.investigation_id,
}
system_prompt_with_truncated_tools = load_and_render_prompt(
template_path, template_context
Expand Down
6 changes: 0 additions & 6 deletions holmes/core/investigation.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,7 +9,6 @@
from holmes.core.supabase_dal import SupabaseDal
from holmes.core.tracing import DummySpan, SpanType
from holmes.utils.global_instructions import add_global_instructions_to_user_prompt
from holmes.core.todo_manager import get_todo_manager

from holmes.core.investigation_structured_output import (
DEFAULT_SECTIONS,
Expand Down Expand Up @@ -133,9 +132,6 @@ def get_investigation_context(
else:
logging.info("Structured output is disabled for this request")

todo_manager = get_todo_manager()
todo_context = todo_manager.format_tasks_for_prompt(ai.investigation_id)

system_prompt = load_and_render_prompt(
investigate_request.prompt_template,
{
Expand All @@ -144,8 +140,6 @@ def get_investigation_context(
"structured_output": request_structured_output_from_llm,
"toolsets": ai.tool_executor.toolsets,
"cluster_name": config.cluster_name,
"todo_list": todo_context,
"investigation_id": ai.investigation_id,
},
)

Expand Down
61 changes: 60 additions & 1 deletion holmes/core/llm.py
Original file line number Diff line number Diff line change
Expand Up @@ -229,9 +229,11 @@ def completion(
] # can be removed after next litelm version

self.args.setdefault("temperature", temperature)

self._add_cache_control_to_last_message(messages)
Comment thread
arikalon1 marked this conversation as resolved.

Comment thread
arikalon1 marked this conversation as resolved.
# Get the litellm module to use (wrapped or unwrapped)
litellm_to_use = self.tracer.wrap_llm(litellm) if self.tracer else litellm

result = litellm_to_use.completion(
model=self.model,
api_key=self.api_key,
Expand Down Expand Up @@ -266,3 +268,60 @@ def get_maximum_output_token(self) -> int:
f"Couldn't find model's name {model_name} in litellm's model list, fallback to 4096 tokens for max_output_tokens"
)
return 4096

def _add_cache_control_to_last_message(
self, messages: List[Dict[str, Any]]
) -> None:
"""
Add cache_control to the last non-user message for Anthropic prompt caching.
Removes any existing cache_control from previous messages to avoid accumulation.
"""
# First, remove any existing cache_control from all messages
for msg in messages:
content = msg.get("content")
if isinstance(content, list):
for block in content:
if isinstance(block, dict) and "cache_control" in block:
del block["cache_control"]
logging.debug(
f"Removed existing cache_control from {msg.get('role')} message"
)

# Find the last non-user message to add cache_control to.
# Adding cache_control to user message requires changing its structure, so we avoid it
# This avoids breaking parse_messages_tags which only processes user messages
target_msg = None
for msg in reversed(messages):
if msg.get("role") != "user":
target_msg = msg
break

if not target_msg:
logging.debug("No non-user message found for cache_control")
return

content = target_msg.get("content")

if content is None:
Comment thread
arikalon1 marked this conversation as resolved.
return

if isinstance(content, str):
# Convert string to structured format with cache_control
target_msg["content"] = [
{
"type": "text",
"text": content,
"cache_control": {"type": "ephemeral"},
}
]
logging.debug(
f"Added cache_control to {target_msg.get('role')} message (converted from string)"
)
elif isinstance(content, list) and content:
# Add cache_control to the last content block
last_block = content[-1]
if isinstance(last_block, dict) and "type" in last_block:
last_block["cache_control"] = {"type": "ephemeral"}
logging.debug(
f"Added cache_control to {target_msg.get('role')} message (structured content)"
)
2 changes: 0 additions & 2 deletions holmes/core/prompt.py
Original file line number Diff line number Diff line change
Expand Up @@ -40,7 +40,6 @@ def build_initial_ask_messages(
initial_user_prompt: str,
file_paths: Optional[List[Path]],
tool_executor: Any, # ToolExecutor type
investigation_id: str,
runbooks: Union[RunbookCatalog, Dict, None] = None,
system_prompt_additions: Optional[str] = None,
) -> List[Dict]:
Expand All @@ -60,7 +59,6 @@ def build_initial_ask_messages(
"toolsets": tool_executor.toolsets,
"runbooks": runbooks or {},
"system_prompt_additions": system_prompt_additions or "",
"investigation_id": investigation_id,
}
system_prompt_rendered = load_and_render_prompt(
system_prompt_template, template_context
Expand Down
88 changes: 0 additions & 88 deletions holmes/core/todo_manager.py

This file was deleted.

Loading
Loading