Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions docs/development/evals/writing.md
Original file line number Diff line number Diff line change
Expand Up @@ -337,8 +337,8 @@ spec:

```yaml
user_prompt: 'How is the test-nginx deployment performing?'
before-test: kubectl apply -f manifest.yaml
after-test: kubectl delete -f manifest.yaml
before_test: kubectl apply -f manifest.yaml
after_test: kubectl delete -f manifest.yaml
# ... rest of configuration
```

Expand Down
9 changes: 7 additions & 2 deletions holmes/core/tools_utils/toolset_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -16,12 +16,17 @@ def filter_out_default_logging_toolset(toolsets: list[Toolset]) -> list[Toolset]
All other types of toolsets are included as is.
"""

logging_toolsets: list[BasePodLoggingToolset] = []
logging_toolsets: list[Toolset] = []
Comment thread
aantn marked this conversation as resolved.
final_toolsets: list[Toolset] = []

for ts in toolsets:
toolset_type = (
ts.original_toolset_type
if hasattr(ts, "original_toolset_type")
else type(ts)
)
if (
isinstance(ts, BasePodLoggingToolset)
issubclass(toolset_type, BasePodLoggingToolset)
and ts.status == ToolsetStatusEnum.ENABLED
):
logging_toolsets.append(ts)
Expand Down
17 changes: 16 additions & 1 deletion poetry.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

1 change: 1 addition & 0 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -76,6 +76,7 @@ tomli = {version = "^2.0.1", python = "<3.11"}
mypy = "^1.16.0"
pytest-cov = "^6.2.1"
types-python-dateutil = "^2.9.0.20250708"
pytest-dotenv = "^0.5.2"

[build-system]
requires = ["poetry-core"]
Expand Down
16 changes: 13 additions & 3 deletions tests/llm/conftest.py
Original file line number Diff line number Diff line change
Expand Up @@ -58,11 +58,21 @@ def mock_generation_config(request):
if regenerate_all_mocks:
generate_mocks = True

run_live = os.getenv("RUN_LIVE", "False").lower() in ("true", "1", "t")
if generate_mocks and not run_live:
print(
"⚠️ WARNING: --generate-mocks is set but RUN_LIVE is not set. This will not generate mocks."
)
pytest.skip(
"Skipping test case because --generate-mocks is set but RUN_LIVE is not set."
)

# Determine mode based on environment and options
if os.getenv("RUN_LIVE", "False").lower() in ("true", "1", "t"):

if generate_mocks:
mode = MockMode.GENERATE # live & generate
elif run_live:
mode = MockMode.LIVE
elif generate_mocks:
mode = MockMode.GENERATE
else:
mode = MockMode.MOCK

Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,7 @@ description: |
and the actual issue is missing tolerations for node taints, not zone constraints.
Tests that LLM identifies the real scheduling constraint.

user_question: "Whats wrong with database-primary deployment in production-88 isn't starting?"
user_prompt: "Whats wrong with database-primary deployment in production-88 isn't starting?"

# CodeRabbit: RWO PVC + 4 replicas ⇒ inevitable Multi-Attach errors
# A single ReadWriteOnce volume cannot be mounted by four pods on different nodes. This scheduling error will surface before the intended taint/toleration failure, potentially breaking the test signal.
Expand Down

Large diffs are not rendered by default.

Original file line number Diff line number Diff line change
@@ -0,0 +1,2 @@
{"toolset_name":"datadog/logs","tool_name":"fetch_pod_logs","match_params":{"pod_name":"catalogue-db-c948fd796-w4vzl","namespace":"sock-shop","start_time":"-3600"}}
{"schema_version": "robusta:v1.0.0", "status": "no_data", "error": null, "return_code": null, "url": null, "invocation": null, "params": {"namespace": "sock-shop", "pod_name": "catalogue-db-c948fd796-w4vzl", "start_time": "-3600", "end_time": null, "filter": null, "limit": null}}

Large diffs are not rendered by default.

Original file line number Diff line number Diff line change
@@ -0,0 +1,23 @@
description: |
A test case that simulates an issue real customer bumped into.
When holmes were asked to fetch logs, they were not able to do so.
The main reason behind this are due to the conversation history and the lack of explicity saying that we enabled datadog logs toolset as part of the system prompt logging section.

user_prompt:
- "Can you retrieve the datadog logs for this?"
- "Please show me the logs for this pod"
- "Show me logs"


expected_output: |
Logs for catalogue-f7687cb4-zsngc pod show consistent health checks with no errors.

evaluation:
correctness: 1

test_type: "server"

# setup & teardown

tags:
- easy
Original file line number Diff line number Diff line change
@@ -0,0 +1,15 @@
toolsets:
prometheus/metrics:
enabled: False
kubernetes/kube-lineage-extras:
enabled: true
kubernetes/logs:
enabled: False
kubernetes/core:
enabled: true
datadog/logs:
enabled: True
datadog/metrics:
enabled: True
datadog/traces:
enabled: True
Original file line number Diff line number Diff line change
@@ -1,9 +1,9 @@
expected_output:
- The frontend-service pod in the default namespace is experiencing 100% CPU throttling for the stress container
- Suggest increasing the CPU limit
before-test: |
before_test: |
kubectl apply -f manifest.yaml
sleep 60
after-test: kubectl delete -f manifest.yaml
after_test: kubectl delete -f manifest.yaml
evaluation:
correctness: 1
4 changes: 3 additions & 1 deletion tests/llm/test_ask_holmes.py
Original file line number Diff line number Diff line change
Expand Up @@ -289,7 +289,9 @@ def ask_holmes(
llm=DefaultLLM(os.environ.get("MODEL", "gpt-4o"), tracer=tracer),
)

test_type = os.environ.get("ASK_HOLMES_TEST_TYPE", "cli").lower()
test_type = (
test_case.test_type or os.environ.get("ASK_HOLMES_TEST_TYPE", "cli").lower()
)
Comment thread
moshemorad marked this conversation as resolved.
if test_type == "cli":
if test_case.conversation_history:
pytest.skip("CLI mode does not support conversation history tests")
Expand Down
Loading