Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
@@ -1,20 +1,4 @@
user_prompt:
- "generate me a graph of robusta-runner"
- "Show me robusta-runner memory usage over the last 24 hours"
- "Graph robusta-runner memory consumption trends for the past week"
- "Display hourly memory usage pattern for robusta-runner over the last 3 days"
- "Graph robusta-runner memory utilization over time"
- "Display robusta-runner memory usage progression over the last week"
- "Show me how robusta-runner memory usage changed during the last day"
- "Graph robusta-runner memory consumption over time"
- "Display robusta-runner memory usage timeline"
- "Show me robusta-runner memory usage trends over the last 7 days"
- "Can you show me a memory graph of the robusta-runner pod?"
- "Show me memory metrics for robusta-runner"
- "Display robusta-runner pod memory usage graph"
- "What's the memory utilization of the robusta-runner deployment?"
- "Generate a memory graph for robusta-runner pods"
- "Show memory charts for robusta-runner"
user_prompt: "Show me robusta-runner CPU usage over the last 24 hours"
expected_output: |
Output must contain 1 or more embeds in the following format
<<{"type": "datadogql", "tool_name": "query_datadog_metrics", "tool_call_id": "iD8G"}>>
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -17,6 +17,10 @@ tags:
- logs
- datadog
- hard

toolsets_matrix:
- toolsets.yaml
- toolsets_http.yaml
# - benchmark
# we need the datadog api key for it to work as a benchmark

Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,45 @@
toolsets:
datadog-logs-api:
type: http
enabled: true
config:
endpoints:
- hosts: ["api.us5.datadoghq.com"]
paths: ["/api/v2/logs/*"]
methods: ["POST"]
auth:
type: header
name: "DD-API-KEY"
value: "{{env.DATADOG_API_KEY}}"
default_headers:
DD-APPLICATION-KEY: "{{env.DATADOG_APP_KEY}}"
timeout_seconds: 30
llm_instructions: |
### Datadog Logs REST API

Base URL: https://api.us5.datadoghq.com

Available endpoints:
- POST /api/v2/logs/events/search - Search logs

Request body format:
{
"filter": {
"query": "search query using Datadog log search syntax",
"from": "unix_epoch_milliseconds_as_string",
"to": "unix_epoch_milliseconds_as_string",
"indexes": ["*"],
"storage_tier": "indexes"
},
"page": {"limit": 100},
"sort": "-timestamp"
}

Time parameters use unix epoch MILLISECONDS as strings (e.g. "1700000000000").
Search syntax examples:
- "pod_name:coral-reef*" - match pods by name prefix
- "service:web-app @http.status_code:500" - filter by service and status
- "@message:*error*" - search in message content
Use wildcards (*) for pattern matching in pod names.

For pagination, use the cursor from the response: add "cursor" to the "page" object.
1 change: 1 addition & 0 deletions tests/llm/test_ask_holmes.py
Original file line number Diff line number Diff line change
Expand Up @@ -189,6 +189,7 @@ def ask_holmes(
mock_policy=test_case.mock_policy,
mock_overrides=test_case.mock_overrides,
allow_toolset_failures=getattr(test_case, "allow_toolset_failures", False),
toolsets_config_path=getattr(test_case, "toolsets_config_path", None),
)

tool_executor = ToolExecutor(toolset_manager.toolsets)
Expand Down
3 changes: 3 additions & 0 deletions tests/llm/test_investigate.py
Original file line number Diff line number Diff line change
Expand Up @@ -60,6 +60,9 @@ def create_tool_executor(self, dal: Optional[SupabaseDal]) -> ToolExecutor:
mock_generation_config=self._mock_generation_config,
mock_policy=self._test_case.mock_policy,
mock_overrides=getattr(self._test_case, "mock_overrides", None),
toolsets_config_path=getattr(
self._test_case, "toolsets_config_path", None
),
)

# With the new file-based mock system, mocks are loaded from disk automatically
Expand Down
7 changes: 6 additions & 1 deletion tests/llm/utils/mock_toolset.py
Original file line number Diff line number Diff line change
Expand Up @@ -603,12 +603,14 @@ def __init__(
mock_policy: str = "inherit",
mock_overrides: Optional[Dict[str, str]] = None,
allow_toolset_failures: bool = False,
toolsets_config_path: Optional[str] = None,
):
self.test_case_folder = test_case_folder
self.request = request
self.mock_overrides = mock_overrides or {}
self.mock_generation_config = mock_generation_config
self.allow_toolset_failures = allow_toolset_failures
self.toolsets_config_path = toolsets_config_path

# Coerce mock_policy string to MockPolicy enum, falling back to INHERIT for unknown values
if isinstance(mock_policy, str):
Expand Down Expand Up @@ -673,7 +675,10 @@ def _initialize_toolsets(self):
builtin_toolsets = load_builtin_toolsets(mock_dal)

# Load custom toolsets from YAML if present
config_path = os.path.join(self.test_case_folder, "toolsets.yaml")
# Use explicit toolsets_config_path (from toolsets_matrix) or fall back to default
config_path = self.toolsets_config_path or os.path.join(
self.test_case_folder, "toolsets.yaml"
)
custom_definitions = self._load_custom_toolsets(config_path)

# Always load default toolsets.yaml
Expand Down
71 changes: 71 additions & 0 deletions tests/llm/utils/test_case_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -146,6 +146,15 @@ class HolmesTestCase(BaseModel):
port_forwards: Optional[List[Dict[str, Any]]] = (
None # Port forwarding configurations
)
toolsets_matrix: Optional[List[str]] = (
None # List of toolset config filenames for matrix expansion
)
toolsets_config_name: Optional[str] = (
None # Derived name of the active toolset config (auto-set during matrix expansion)
)
toolsets_config_path: Optional[str] = (
None # Full path to the active toolset config file (auto-set during matrix expansion)
)


class AskHolmesTestCase(HolmesTestCase, BaseModel):
Expand Down Expand Up @@ -254,6 +263,64 @@ def _add_port_forward_tag(self, test_case: HolmesTestCase) -> None:
if "port-forward" not in test_case.tags:
test_case.tags.append("port-forward")

@staticmethod
def _derive_toolset_config_name(filename: str) -> str:
"""Derive a short name from a toolset config filename for use in test IDs.

Examples:
toolsets_builtin.yaml -> builtin
toolsets_http_datadog.yaml -> http_datadog
toolsets.yaml -> default
custom.yaml -> custom
"""
name = filename
for ext in (".yaml", ".yml"):
if name.endswith(ext):
name = name[: -len(ext)]
break
if name.startswith("toolsets_"):
name = name[len("toolsets_") :]
elif name == "toolsets":
name = "default"
return name or "default"

def _expand_toolsets_matrix(
self, test_cases: List[HolmesTestCase]
) -> List[HolmesTestCase]:
"""Expand test cases that have toolsets_matrix into multiple variants.

Each entry in toolsets_matrix is a filename (e.g. toolsets_builtin.yaml)
that must exist in the test case folder. For each file, a variant of the
test case is created with toolsets_config_name and toolsets_config_path set.
The variant ID is appended with [config_name].
"""
expanded: List[HolmesTestCase] = []
for tc in test_cases:
if not tc.toolsets_matrix:
expanded.append(tc)
continue

for config_filename in tc.toolsets_matrix:
config_path = os.path.join(tc.folder, config_filename)
if not os.path.isfile(config_path):
raise FileNotFoundError(
f"Toolsets matrix config file '{config_filename}' not found "
f"in test case folder: {tc.folder}"
)

name = self._derive_toolset_config_name(config_filename)

variant = tc.model_copy(deep=True)
variant.toolsets_config_name = name
variant.toolsets_config_path = config_path
variant.id = f"{tc.id}[{name}]"
if not variant.base_id:
variant.base_id = tc.id

expanded.append(variant)

return expanded

def load_test_cases(self) -> List[HolmesTestCase]:
test_cases: List[HolmesTestCase] = []
test_cases_ids: List[str] = [
Expand Down Expand Up @@ -394,6 +461,10 @@ def load_test_cases(self) -> List[HolmesTestCase]:
continue
logging.debug(f"Found {len(test_cases)} in {self._test_cases_folder}")

# Expand toolsets_matrix variants (must happen after all test cases are loaded,
# including array prompt expansion, to produce the cross-product)
test_cases = self._expand_toolsets_matrix(test_cases)

return test_cases


Expand Down
Loading