Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
37 changes: 36 additions & 1 deletion holmes/core/llm.py
Original file line number Diff line number Diff line change
Expand Up @@ -347,6 +347,8 @@ def completion(
temperature: Optional[float] = None,
drop_params: Optional[bool] = None,
stream: Optional[bool] = None,
user: Optional[str] = None,
metadata: Optional[Dict[str, Any]] = None,
) -> Union[ModelResponse, CustomStreamWrapper]:
pass

Expand Down Expand Up @@ -654,6 +656,8 @@ def completion(
temperature: Optional[float] = None,
drop_params: Optional[bool] = None,
stream: Optional[bool] = None,
user: Optional[str] = None,
metadata: Optional[Dict[str, Any]] = None,
) -> Union[ModelResponse, CustomStreamWrapper]:
tools_args = {}
allowed_openai_params = None
Expand Down Expand Up @@ -752,6 +756,34 @@ def completion(
}
]

# Provider-neutral trace attribution. `user` is the standard end-user
# identifier; `metadata` carries optional observability fields (session
# id, tags) that are used only for logging and never sent to the model.
# Both are forwarded as-is to the underlying LLM client so any configured
# observability backend can group and filter traces by user/conversation,
# whichever model provider serves the request. Explicit call arguments
# win over any statically-configured values in self.args (metadata is
# merged key-by-key; user is replaced).
#
# self.args is read non-destructively (and `user`/`metadata` are excluded
# from the spread below) so a reused DefaultLLM keeps its configured
# values across calls — completion() runs once per call_stream iteration.
attribution_kwargs: Dict[str, Any] = {}
configured_user = self.args.get("user")
effective_user = user if user is not None else configured_user
if effective_user is not None:
attribution_kwargs["user"] = effective_user

configured_metadata = self.args.get("metadata")
if configured_metadata or metadata:
merged_metadata: Dict[str, Any] = {}
if isinstance(configured_metadata, dict):
merged_metadata.update(configured_metadata)
if metadata:
merged_metadata.update(metadata)
if merged_metadata:
attribution_kwargs["metadata"] = merged_metadata

result = litellm_to_use.completion(
model=litellm_model_name,
api_key=self.api_key,
Expand All @@ -765,8 +797,11 @@ def completion(
timeout=LLM_REQUEST_TIMEOUT,
**azure_ad_kwargs,
**tools_args,
**self.args,
# `user`/`metadata` are handled via attribution_kwargs; exclude them
# here so they are never passed twice.
**{k: v for k, v in self.args.items() if k not in ("user", "metadata")},
**cache_kwargs,
**attribution_kwargs,
)

if isinstance(result, ModelResponse):
Expand Down
101 changes: 101 additions & 0 deletions holmes/core/llm_observability.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,101 @@
"""Derive end-user / session attribution for an LLM call from a request.

Holmes runs as a server on behalf of many users. Attaching the end-user
identifier and the conversation/session to each LLM call lets any observability
backend group and filter traces by user and conversation — regardless of which
model provider (OpenAI, Anthropic, Bedrock, ...) actually serves the request.

Attribution is expressed with provider-neutral fields:

- ``user`` is the standard end-user identifier understood across providers.
Unlike ``metadata``, it is forwarded all the way to the model provider (see
``DefaultLLM.completion()``), so it is a stable hash of the identifier rather
than the raw value — the same guidance OpenAI gives for this field ("we
recommend hashing their username or email to avoid sending us any
identifying information"). The hash is deterministic, so a given user still
maps to a single, filterable trace identity.
- ``metadata`` carries additional, optional observability fields (session id,
tags) that Holmes forwards without interpreting them. Unlike ``user``, this
is a logging-only field consumed by whichever process makes the call — it is
not sent to the model provider — so it does not need hashing here.

This module is the single, deliberately narrow place that decides *what* of an
inbound request becomes attribution data. It is a whitelist on purpose: only
known-safe identity fields are mapped, and free-form tags are bounded, so nothing
arbitrary from ``request_context`` leaks into traces.
"""

import hashlib
from dataclasses import dataclass
from typing import Any, Dict, List, Optional

# Defensive bounds so a misbehaving caller cannot blow up trace cardinality.
_MAX_TAGS = 20
_MAX_TAG_LEN = 256


@dataclass(frozen=True)
class TraceAttribution:
"""Provider-neutral attribution for a single LLM call."""

user: Optional[str] = None
metadata: Optional[Dict[str, Any]] = None

def is_empty(self) -> bool:
return self.user is None and not self.metadata


def _clean(value: Any) -> Optional[str]:
if value is None:
return None
text = str(value).strip()
return text or None


def _hash_identifier(value: str) -> str:
"""Stable, one-way hash for a value forwarded to the model provider.

Deterministic (same input always yields the same output) so traces from
the same user still group together, without exposing the raw identifier
to whichever provider serves the request.
"""
return hashlib.sha256(value.encode("utf-8")).hexdigest()


def build_trace_attribution(
request_context: Optional[Dict[str, Any]],
) -> TraceAttribution:
"""Build provider-neutral trace attribution from a request context.

- ``user`` ← sha256(``user_email`` (preferred) or ``user_id``)
- ``session_id`` (metadata) ← ``conversation_id``
- ``tags`` (metadata) ← ``request_type:<...>`` and ``cluster:<...>``

Returns an empty :class:`TraceAttribution` when there is nothing to
attribute, so behaviour is unchanged for callers that carry no identity
(e.g. the CLI).
"""
if not request_context:
return TraceAttribution()

raw_user = _clean(request_context.get("user_email")) or _clean(
request_context.get("user_id")
)
user = _hash_identifier(raw_user) if raw_user else None

metadata: Dict[str, Any] = {}
session = _clean(request_context.get("conversation_id"))
if session:
metadata["session_id"] = session

tags: List[str] = []
request_type = _clean(request_context.get("request_type"))
if request_type:
tags.append(f"request_type:{request_type}")
cluster_name = _clean(request_context.get("cluster_name"))
if cluster_name:
tags.append(f"cluster:{cluster_name}")
if tags:
metadata["tags"] = [tag[:_MAX_TAG_LEN] for tag in tags[:_MAX_TAGS]]

return TraceAttribution(user=user, metadata=metadata or None)
4 changes: 4 additions & 0 deletions holmes/core/tool_calling_llm.py
Original file line number Diff line number Diff line change
Expand Up @@ -26,6 +26,7 @@
load_bool,
)
from holmes.core.llm import LLM
from holmes.core.llm_observability import build_trace_attribution
from holmes.core.llm_usage import RequestStats
from holmes.core.models import (
FrontendToolResult,
Expand Down Expand Up @@ -1160,6 +1161,7 @@ def call_stream(
with trace_span.start_span(name="gen_ai.chat") as llm_span:
try:
_llm_call_start = time.time()
attribution = build_trace_attribution(self._request_context)
full_response = self.llm.completion(
messages=parse_messages_tags(messages), # type: ignore
tools=tools,
Expand All @@ -1168,6 +1170,8 @@ def call_stream(
temperature=TEMPERATURE,
stream=False,
drop_params=True,
user=attribution.user,
metadata=attribution.metadata,
)

# Accumulate cost information for this iteration
Expand Down
7 changes: 7 additions & 0 deletions server.py
Original file line number Diff line number Diff line change
Expand Up @@ -528,6 +528,13 @@ def chat(chat_request: ChatRequest, http_request: Request):
if chat_request.user_id:
request_context.setdefault("headers", {})
request_context["user_id"] = chat_request.user_id
# user_email is the frontend-supplied source of truth for usage
# analytics; surface it so LLM-call observability metadata can attribute
# traces to the end user (see holmes.core.llm_observability).
if chat_request.user_email:
request_context["user_email"] = chat_request.user_email
if chat_request.request_type:
request_context["request_type"] = chat_request.request_type
# Surface conversation_id and cluster_name to toolsets that need
# to hardwire them into outbound requests (e.g. platform-mcp adds
# them as X-Robusta-* headers so tool handlers don't have to trust
Expand Down
121 changes: 121 additions & 0 deletions tests/core/test_llm_completion_metadata.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,121 @@
from unittest.mock import patch

import pytest
from litellm.types.utils import Choices, Message, ModelResponse, Usage

from holmes.core.llm import DefaultLLM


def _mock_model_response() -> ModelResponse:
return ModelResponse(
id="chatcmpl-test",
choices=[
Choices(
index=0,
message=Message(role="assistant", content="ok", tool_calls=None),
finish_reason="stop",
)
],
model="test-model",
usage=Usage(prompt_tokens=1, completion_tokens=1, total_tokens=2),
)


def _make_llm(args: dict | None = None) -> DefaultLLM:
"""Build a DefaultLLM bypassing __init__/check_llm so we can control self.args."""
llm = DefaultLLM.__new__(DefaultLLM)
llm.model = "openai/Claude Sonnet 4.6"
llm.api_key = None
llm.api_base = None
llm.api_version = None
llm.args = args or {}
llm.tracer = None
llm.name = None
llm.is_robusta_model = False
llm.max_context_size = None
return llm


@pytest.fixture
def mock_completion():
with patch("holmes.core.llm.litellm.completion") as mock:
mock.return_value = _mock_model_response()
yield mock


class TestCompletionAttribution:
"""`user` is the standard, provider-neutral end-user identifier; `metadata`
carries optional observability fields. completion() must forward both so the
configured observability backend can attribute traces to the end user."""

def test_user_is_forwarded(self, mock_completion):
llm = _make_llm()
llm.completion(
messages=[{"role": "user", "content": "hi"}], user="alice@example.com"
)
assert mock_completion.call_args.kwargs.get("user") == "alice@example.com"

def test_metadata_is_forwarded(self, mock_completion):
llm = _make_llm()
md = {"session_id": "conv-1", "tags": ["request_type:user_chat"]}
llm.completion(messages=[{"role": "user", "content": "hi"}], metadata=md)
assert mock_completion.call_args.kwargs.get("metadata") == md

def test_no_attribution_means_no_kwargs(self, mock_completion):
"""Default behaviour is unchanged: neither kwarg is sent."""
llm = _make_llm()
llm.completion(messages=[{"role": "user", "content": "hi"}])
kwargs = mock_completion.call_args.kwargs
assert "user" not in kwargs
assert "metadata" not in kwargs

def test_none_attribution_means_no_kwargs(self, mock_completion):
llm = _make_llm()
llm.completion(
messages=[{"role": "user", "content": "hi"}], user=None, metadata=None
)
kwargs = mock_completion.call_args.kwargs
assert "user" not in kwargs
assert "metadata" not in kwargs

def test_per_call_user_overrides_configured(self, mock_completion):
llm = _make_llm({"user": "static-user"})
llm.completion(messages=[{"role": "user", "content": "hi"}], user="alice")
assert mock_completion.call_args.kwargs.get("user") == "alice"

def test_per_call_metadata_merges_over_configured(self, mock_completion):
"""Statically-configured metadata is merged with per-call metadata,
per-call keys winning on conflict."""
llm = _make_llm({"metadata": {"session_id": "static", "fixed": "keep"}})
llm.completion(
messages=[{"role": "user", "content": "hi"}],
metadata={"session_id": "s-1", "tags": ["t"]},
)
assert mock_completion.call_args.kwargs.get("metadata") == {
"session_id": "s-1",
"fixed": "keep",
"tags": ["t"],
}

def test_configured_values_are_not_passed_twice(self, mock_completion):
"""Configured user/metadata must be excluded from the **self.args spread
so they are not also passed as duplicate kwargs (which would raise)."""
llm = _make_llm({"user": "u", "metadata": {"fixed": "keep"}})
llm.completion(messages=[{"role": "user", "content": "hi"}])
kwargs = mock_completion.call_args.kwargs
assert kwargs.get("user") == "u"
assert kwargs.get("metadata") == {"fixed": "keep"}

def test_configured_values_survive_repeated_calls(self, mock_completion):
"""A reused DefaultLLM must keep its configured user/metadata across
calls — self.args is read non-destructively, not popped."""
llm = _make_llm({"user": "u", "metadata": {"fixed": "keep"}})
for _ in range(2):
llm.completion(messages=[{"role": "user", "content": "hi"}])
kwargs = mock_completion.call_args.kwargs # last (2nd) call
assert kwargs.get("user") == "u"
assert kwargs.get("metadata") == {"fixed": "keep"}
# completion() does not consume the configured attribution from self.args
# (it reads them non-destructively, so they persist across calls).
assert llm.args.get("user") == "u"
assert llm.args.get("metadata") == {"fixed": "keep"}
Loading