Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
51 changes: 51 additions & 0 deletions docs/data-sources/remote-mcp-servers.md
Original file line number Diff line number Diff line change
Expand Up @@ -46,6 +46,57 @@ mcp_servers:
llm_instructions: "This server provides general data access capabilities. Use it when you need to retrieve external information or perform remote operations that aren't covered by other toolsets."
```

#### Dynamic Headers with Request Context

MCP servers can use dynamic headers that are populated from the incoming HTTP request context. This is useful for passing authentication tokens or other request-specific headers to your MCP server.

Use the `extra_headers` field (instead of `headers`) with template variables to reference headers from the incoming request:

```yaml
mcp_servers:
my_server:
description: "My MCP server with dynamic authentication"
config:
url: "http://example.com:8000/mcp/messages"
mode: streamable-http
extra_headers:
X-Auth-Token: "{{ request_context.headers['X-Auth-Token'] }}"
X-User-Id: "{{ request_context.headers['X-User-Id'] }}"
llm_instructions: "Use this server to access resources with per-request authentication."
```

**How it works:**

- When a request comes to HolmesGPT (via the server API), headers from that request are available in `request_context.headers`
- Header lookups are case-insensitive (e.g., `X-Auth-Token`, `x-auth-token`, and `X-AUTH-TOKEN` all work)
- The template is rendered when calling the MCP server, passing the header value through
- You can also use environment variables: `"{{ env.MY_VAR }}"` or combine them: `"Bearer {{ request_context.headers['token'] }}"`
Comment thread
gossion marked this conversation as resolved.

**Example use case:**

This is particularly useful when your MCP server needs to authenticate with external services using tokens that are specific to each request/user.

```yaml
mcp_servers:
remote_api_server:
description: "Remote API MCP Server"
config:
url: "http://mcp-server:8000/mcp"
mode: streamable-http
extra_headers:
X-Auth-Token: "{{ request_context.headers['X-Auth-Token'] }}"
llm_instructions: "Use this server to interact with remote APIs."
```

When making requests to HolmesGPT, include the required header:

```bash
curl -X POST http://holmes-server/api/investigate \
-H "X-Auth-Token: your-auth-token-here" \
-H "Content-Type: application/json" \
-d '{"question": "Check system status"}'
Comment thread
arikalon1 marked this conversation as resolved.
```

### URL Format

The URL should point to the MCP server endpoint. The exact path depends on your server configuration:
Expand Down
4 changes: 3 additions & 1 deletion holmes/core/investigation.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@
import logging
from typing import Optional
from typing import Any, Dict, Optional

from holmes.config import Config
from holmes.core.investigation_structured_output import (
Expand All @@ -26,6 +26,7 @@ def investigate_issues(
model: Optional[str] = None,
trace_span=DummySpan(),
runbooks: Optional[RunbookCatalog] = None,
request_context: Optional[Dict[str, Any]] = None,
) -> InvestigationResult:
context = dal.get_issue_data(investigate_request.context.get("robusta_issue_id"))

Expand Down Expand Up @@ -57,6 +58,7 @@ def investigate_issues(
sections=investigate_request.sections,
trace_span=trace_span,
runbooks=runbooks,
request_context=request_context,
)

(text_response, sections) = process_response_into_sections(investigation.result)
Expand Down
33 changes: 30 additions & 3 deletions holmes/core/tool_calling_llm.py
Original file line number Diff line number Diff line change
Expand Up @@ -258,7 +258,10 @@ def reset_interaction_state(self) -> None:
self._runbook_in_use = False

def process_tool_decisions(
self, messages: List[Dict[str, Any]], tool_decisions: List[ToolApprovalDecision]
self,
messages: List[Dict[str, Any]],
tool_decisions: List[ToolApprovalDecision],
request_context: Optional[Dict[str, Any]] = None,
) -> tuple[List[Dict[str, Any]], list[StreamMessage]]:
"""
Process tool approval decisions and execute approved tools.
Expand Down Expand Up @@ -319,6 +322,7 @@ def process_tool_decisions(
tool_number=None,
user_approved=True,
session_approved_prefixes=session_prefixes,
request_context=request_context,
)
else:
# Tool was rejected or no decision found, add rejection message
Expand Down Expand Up @@ -369,6 +373,7 @@ def prompt_call(
response_format: Optional[Union[dict, Type[BaseModel]]] = None,
sections: Optional[InputSectionsDataType] = None,
trace_span=DummySpan(),
request_context: Optional[Dict[str, Any]] = None,
) -> LLMResult:
messages = [
{"role": "system", "content": system_prompt},
Expand All @@ -380,16 +385,21 @@ def prompt_call(
user_prompt=user_prompt,
sections=sections,
trace_span=trace_span,
request_context=request_context,
)

def messages_call(
self,
messages: List[Dict[str, str]],
response_format: Optional[Union[dict, Type[BaseModel]]] = None,
trace_span=DummySpan(),
request_context: Optional[Dict[str, Any]] = None,
) -> LLMResult:
return self.call(
messages, response_format=response_format, trace_span=trace_span
messages,
response_format=response_format,
trace_span=trace_span,
request_context=request_context,
)

def _should_include_restricted_tools(self) -> bool:
Expand All @@ -412,6 +422,7 @@ def call( # type: ignore
sections: Optional[InputSectionsDataType] = None,
trace_span=DummySpan(),
tool_number_offset: int = 0,
request_context: Optional[Dict[str, Any]] = None,
) -> LLMResult:
tool_calls: list[
dict
Expand Down Expand Up @@ -546,6 +557,7 @@ def call( # type: ignore
previous_tool_calls=tool_calls,
trace_span=trace_span,
tool_number=tool_number,
request_context=request_context,
)
futures_tool_numbers[future] = tool_number
futures.append(future)
Expand All @@ -567,6 +579,7 @@ def call( # type: ignore
tool_call_result=tool_call_result,
tool_number=tool_number,
trace_span=trace_span,
request_context=request_context,
)

tool_result_response_dict = (
Expand Down Expand Up @@ -603,6 +616,7 @@ def _directly_invoke_tool_call(
tool_call_id: str,
tool_number: Optional[int] = None,
session_approved_prefixes: Optional[List[str]] = None,
request_context: Optional[Dict[str, Any]] = None,
) -> StructuredToolResult:
tool = self.tool_executor.get_tool_by_name(tool_name)
if not tool:
Expand All @@ -624,6 +638,7 @@ def _directly_invoke_tool_call(
tool_name=tool_name,
tool_call_id=tool_call_id,
session_approved_prefixes=session_approved_prefixes or [],
request_context=request_context,
)
tool_response = tool.invoke(tool_params, context=invoke_context)

Expand Down Expand Up @@ -656,6 +671,7 @@ def _get_tool_call_result(
previous_tool_calls: list[dict],
tool_number: Optional[int] = None,
session_approved_prefixes: Optional[List[str]] = None,
request_context: Optional[Dict[str, Any]] = None,
) -> ToolCallResult:
tool_params = {}
try:
Expand All @@ -681,6 +697,7 @@ def _get_tool_call_result(
tool_number=tool_number,
tool_call_id=tool_call_id,
session_approved_prefixes=session_approved_prefixes,
request_context=request_context,
)

if not isinstance(tool_response, StructuredToolResult):
Expand Down Expand Up @@ -750,6 +767,7 @@ def _invoke_llm_tool_call(
tool_number=None,
user_approved: bool = False,
session_approved_prefixes: Optional[List[str]] = None,
request_context: Optional[Dict[str, Any]] = None,
) -> ToolCallResult:
if trace_span is None:
trace_span = DummySpan()
Expand Down Expand Up @@ -784,6 +802,7 @@ def _invoke_llm_tool_call(
tool_number=tool_number,
user_approved=user_approved,
session_approved_prefixes=session_approved_prefixes,
request_context=request_context,
)

original_token_count = prevent_overly_big_tool_response(
Expand Down Expand Up @@ -818,6 +837,7 @@ def _handle_tool_call_approval(
tool_call_result: ToolCallResult,
tool_number: Optional[int],
trace_span: Any,
request_context: Optional[Dict[str, Any]] = None,
) -> ToolCallResult:
"""
Handle approval for a single tool call if required.
Expand Down Expand Up @@ -869,6 +889,7 @@ def _handle_tool_call_approval(
user_approved=True,
tool_number=tool_number,
tool_call_id=tool_call_result.tool_call_id,
request_context=request_context,
)
tool_call_result.result = new_response
else:
Expand All @@ -891,6 +912,7 @@ def call_stream(
msgs: Optional[list[dict]] = None,
enable_tool_approval: bool = False,
tool_decisions: List[ToolApprovalDecision] | None = None,
request_context: Optional[Dict[str, Any]] = None,
):
"""
This function DOES NOT call llm.completion(stream=true).
Expand All @@ -899,7 +921,9 @@ def call_stream(
# Process tool decisions if provided
if msgs and tool_decisions:
logging.info(f"Processing {len(tool_decisions)} tool decisions")
msgs, events = self.process_tool_decisions(msgs, tool_decisions)
msgs, events = self.process_tool_decisions(
msgs, tool_decisions, request_context
)
yield from events

messages: list[dict] = []
Expand Down Expand Up @@ -1040,6 +1064,7 @@ def call_stream(
trace_span=DummySpan(), # Streaming mode doesn't support tracing yet
tool_number=tool_number,
session_approved_prefixes=session_prefixes,
request_context=request_context,
)
futures.append(future)
yield StreamMessage(
Expand Down Expand Up @@ -1179,6 +1204,7 @@ def investigate(
sections: Optional[InputSectionsDataType] = None,
trace_span=DummySpan(),
runbooks: Optional[RunbookCatalog] = None,
request_context: Optional[Dict[str, Any]] = None,
) -> LLMResult:
issue_runbooks = self.runbook_manager.get_instructions_for_issue(issue)

Expand Down Expand Up @@ -1252,6 +1278,7 @@ def investigate(
response_format=response_format,
sections=sections,
trace_span=trace_span,
request_context=request_context,
)
res.instructions = issue_runbooks
return res
16 changes: 16 additions & 0 deletions holmes/core/tools.py
Original file line number Diff line number Diff line change
Expand Up @@ -170,6 +170,22 @@ class ToolInvokeContext(BaseModel):
session_approved_prefixes: List[
str
] = [] # Bash prefixes approved during this session
request_context: Optional[Dict[str, Any]] = None

def model_dump(self, **kwargs):
Comment thread
gossion marked this conversation as resolved.
"""Override to exclude sensitive context from serialization"""
data = super().model_dump(**kwargs)
if data.get("request_context"):
# Sanitize: show keys but not values
data["request_context"] = {
k: "***REDACTED***" for k in data["request_context"].keys()
}
return data
Comment thread
coderabbitai[bot] marked this conversation as resolved.

def __str__(self):
"""Override to prevent accidental context leakage in logs"""
context_keys = list((self.request_context or {}).keys())
return f"ToolInvokeContext(tool_number={self.tool_number}, user_approved={self.user_approved}, context_keys={context_keys})"


class Tool(ABC, BaseModel):
Expand Down
Loading
Loading