Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
54 changes: 53 additions & 1 deletion CLAUDE.md
Original file line number Diff line number Diff line change
Expand Up @@ -118,10 +118,62 @@ export RUN_LIVE=true
poetry run pytest tests/llm/test_ask_holmes.py

# Test with different models
export MODEL=anthropic/claude-3.5
export MODEL=anthropic/claude-3.5-sonnet-20241022
poetry run pytest tests/llm/test_ask_holmes.py
```

### Evaluation CLI Reference

**Custom Pytest Flags**:
- `--generate-mocks`: Generate mock data files during test execution
- `--regenerate-all-mocks`: Regenerate all mock files (implies --generate-mocks)
- `--skip-setup`: Skip before_test commands (useful for iterative testing)
- `--skip-cleanup`: Skip after_test commands (useful for debugging)

**Environment Variables**:
- `MODEL`: LLM model to use (e.g., `gpt-4o`, `anthropic/claude-3-5-sonnet-20241022`)
- `CLASSIFIER_MODEL`: Model for scoring answers (defaults to MODEL)
- `RUN_LIVE=true`: Execute real commands instead of using mocks
- `ITERATIONS=<number>`: Run each test multiple times
- `UPLOAD_DATASET=true`: Sync dataset to Braintrust
- `EXPERIMENT_ID`: Custom experiment name for tracking
- `BRAINTRUST_API_KEY`: Enable Braintrust integration

**Common Evaluation Patterns**:

```bash

# Generate/update mocks for specific tests
poetry run pytest tests/llm/test_ask_holmes.py -k "test_name" --generate-mocks

# Run tests multiple times for reliability
ITERATIONS=100 poetry run pytest tests/llm/test_ask_holmes.py -k "flaky_test"

# Model comparison workflow
EXPERIMENT_ID=gpt4o_baseline MODEL=gpt-4o poetry run pytest tests/llm/ -n 6
EXPERIMENT_ID=claude35_test MODEL=anthropic/claude-3-5-sonnet-20241022 poetry run pytest tests/llm/ -n 6

# Debug with verbose output
poetry run pytest -vv -s tests/llm/test_ask_holmes.py -k "failing_test" --no-cov

# List tests by marker
poetry run pytest -m "llm and not network" --collect-only -q
```

**Available Test Markers**:
- `llm`: LLM behavior tests
- `datetime`: Datetime functionality
- `logs`: Log processing
- `context_window`: Context window handling
- `synthetic`: Synthetic data tests
- `network`: Network-dependent tests
- `runbooks`: Runbook functionality
- `misleading-history`: Misleading data scenarios
- `k8s-misconfig`: Kubernetes misconfigurations
- `chain-of-causation`: Causation analysis
- `slackbot`: Slack integration
- `counting`: Resource counting tests

**Test Infrastructure Notes**:
- All test state tracking uses pytest's `user_properties` to ensure compatibility with pytest-xdist parallel execution
- Mock file tracking and test results are stored in `user_properties` and aggregated in the terminal summary
Expand Down
89 changes: 89 additions & 0 deletions conftest.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,89 @@
import logging
from tests.llm.conftest import show_llm_summary_report


def pytest_addoption(parser):
"""Add custom pytest command line options"""
parser.addoption(
"--generate-mocks",
action="store_true",
default=False,
help="Generate mock data files during test execution instead of using existing mocks",
)
parser.addoption(
"--regenerate-all-mocks",
action="store_true",
default=False,
help="Regenerate all mock data files, replacing existing ones (implies --generate-mocks)",
)
parser.addoption(
"--skip-setup",
action="store_true",
default=False,
help="Skip running before_test commands for test cases (useful for iterative test development)",
)
parser.addoption(
"--skip-cleanup",
action="store_true",
default=False,
help="Skip running after_test commands for test cases (useful for debugging test failures)",
)


def pytest_configure(config):
"""Configure pytest settings"""
# Configure worker-specific log files for xdist compatibility
# worker_id = getattr(config, "workerinput", {}).get("workerid", "master")
# if worker_id != "master":
# # Set worker-specific log file to avoid conflicts
# config.option.log_file = f"tests-{worker_id}.log"

# Determine worker id
# Also see: https://pytest-xdist.readthedocs.io/en/latest/how-to.html#creating-one-log-file-for-each-worker
# worker_id = os.environ.get("PYTEST_XDIST_WORKER", default="gw0")

# # Create logs folder
# logs_folder = os.environ.get("LOGS_FOLDER", default="logs_folder")
# os.makedirs(logs_folder, exist_ok=True)

# # Create file handler to output logs into corresponding worker file
# file_handler = logging.FileHandler(f"{logs_folder}/logs_worker_{worker_id}.log", mode="w")
# file_handler.setFormatter(
# logging.Formatter(
# fmt="{asctime} {levelname}:{name}:{lineno}:{message}",
# style="{",
# )
# )
# # Create stream handler to output logs on console
# # This is a workaround for a known limitation:
# # https://pytest-xdist.readthedocs.io/en/latest/known-limitations.html
# console_handler = logging.StreamHandler(sys.stderr) # pytest only prints error logs
# console_handler.setFormatter(
# logging.Formatter(
# # Include worker id in log messages, \r is needed to separate lines in console
# fmt="\r{asctime} " + worker_id + ":{levelname}:{name}:{lineno}:{message}",
# style="{",
# )
# )
# # Configure logging
# logging.basicConfig(level=logging.INFO, force=True, handlers=[console_handler, file_handler])

# Suppress noisy LiteLLM logs during testing
logging.getLogger("LiteLLM").setLevel(logging.ERROR)
# Also suppress the verbose logger used by LiteLLM
logging.getLogger("LiteLLM.verbose_logger").setLevel(logging.ERROR)
# Suppress litellm sub-loggers
logging.getLogger("litellm").setLevel(logging.ERROR)
logging.getLogger("litellm.cost_calculator").setLevel(logging.ERROR)
logging.getLogger("litellm.litellm_core_utils").setLevel(logging.ERROR)
logging.getLogger("litellm.litellm_core_utils.litellm_logging").setLevel(
logging.ERROR
)
# Suppress httpx HTTP request logs
logging.getLogger("httpx").setLevel(logging.WARNING)
logging.getLogger("httpcore").setLevel(logging.WARNING)


# due to pytest quirks, we need to define this in the main conftest.py - when defined in the llm conftest.py it
# is SOMETIMES picked up and sometimes not, depending on how the test was invokedr
pytest_terminal_summary = show_llm_summary_report
97 changes: 97 additions & 0 deletions docs/development/evals/index.md
Original file line number Diff line number Diff line change
Expand Up @@ -88,6 +88,59 @@ poetry run pytest ./tests/llm/test_ask_holmes.py -k "01_how_many_pods" --no-cov

> It is possible to investigate and debug why an eval fails by the output provided in the console. The output includes the correctness score, the reasoning for the score, information about what tools were called, the expected answer, as well as the LLM's answer.

### Custom Evaluation Flags

HolmesGPT provides custom pytest flags for evaluation workflows:

| Flag | Description | Usage |
|------|-------------|-------|
| `--generate-mocks` | Generate mock data files during test execution | Use when adding new tests or updating existing ones |
| `--regenerate-all-mocks` | Regenerate all mock files (implies --generate-mocks) | Use to ensure mock consistency across all tests |
| `--skip-setup` | Skip `before_test` commands | Use for faster iteration during development |
| `--skip-cleanup` | Skip `after_test` commands | Use for debugging test failures |

### Common Evaluation Patterns

#### Rapid Test Development Workflow

When developing or debugging tests, use this pattern for faster iteration:

```bash
# 1. Initial run with setup (skip cleanup to keep resources)
poetry run pytest tests/llm/test_ask_holmes.py -k "specific_test" --skip-cleanup

# 2. Quick iterations without setup/cleanup
poetry run pytest tests/llm/test_ask_holmes.py -k "specific_test" --skip-setup --skip-cleanup

# 3. Final cleanup when done
poetry run pytest tests/llm/test_ask_holmes.py -k "specific_test" --skip-setup
```

#### Mock Generation

```bash
# Generate mocks for specific test
poetry run pytest tests/llm/test_ask_holmes.py -k "test_name" --generate-mocks

# Generate mocks with multiple iterations to cover all investigative paths
ITERATIONS=100 poetry run pytest tests/llm/test_ask_holmes.py -k "test_name" --generate-mocks

# Regenerate all mocks for consistency
poetry run pytest tests/llm/ --regenerate-all-mocks
```

#### Parallel Execution

For faster test runs, use pytest's parallel execution:

```bash
# Run with 6 parallel workers
poetry run pytest tests/llm/ -n 6 --no-cov --disable-warnings

# Run with auto-detected worker count
poetry run pytest tests/llm/ -n auto --no-cov --disable-warnings
```

### Environment Variables

Configure evaluations using these environment variables:
Expand Down Expand Up @@ -138,6 +191,35 @@ Live testing requires a Kubernetes cluster and will execute `before-test` and `a

3. **Compare Results**: Use evaluation tracking tools to analyze performance differences

## Test Markers

Filter tests using pytest markers:

```bash
# Run only LLM tests
poetry run pytest -m "llm"

# Run tests that don't require network
poetry run pytest -m "not network"

# Combine markers
poetry run pytest -m "llm and not synthetic"
```

**Available markers:**
- `llm` - LLM behavior tests
- `datetime` - Datetime functionality tests
- `logs` - Log processing tests
- `context_window` - Context window handling tests
- `synthetic` - Tests using synthetic data
- `network` - Tests requiring network connectivity
- `runbooks` - Runbook functionality tests
- `misleading-history` - Tests with misleading historical data
- `k8s-misconfig` - Kubernetes misconfiguration tests
- `chain-of-causation` - Chain of causation analysis tests
- `slackbot` - Slack integration tests
- `counting` - Resource counting tests

## Troubleshooting

### Common Issues
Expand All @@ -159,3 +241,18 @@ This shows detailed output including:
- Tool calls made by the LLM
- Evaluation scores and rationales
- Debugging information

### Common Pytest Flags

| Flag | Description |
|------|--------------|
| `-n <number>` | Run tests in parallel with specified workers |
| `-k <pattern>` | Run tests matching the pattern |
| `-m <marker>` | Run tests with specific marker |
| `-v/-vv` | Verbose output (more v's = more verbose) |
| `-s` | Show print statements |
| `--no-cov` | Disable coverage reporting |
| `--disable-warnings` | Disable warning summary |
| `--collect-only` | List tests without running |
| `-q` | Quiet mode |
| `--timeout=<seconds>` | Set test timeout |
23 changes: 23 additions & 0 deletions docs/development/evals/writing.md
Original file line number Diff line number Diff line change
Expand Up @@ -287,6 +287,29 @@ pytest ./tests/llm/test_ask_holmes.py -k "your_test" --generate-mocks

# Or regenerate ALL mocks to ensure consistency
pytest ./tests/llm/test_ask_holmes.py -k "your_test" --regenerate-all-mocks

# Skip setup/cleanup for faster debugging
pytest ./tests/llm/test_ask_holmes.py -k "your_test" --skip-setup --skip-cleanup

# Run with specific number of iterations
ITERATIONS=10 pytest ./tests/llm/test_ask_holmes.py -k "your_test"
```

### CLI Flags Reference

**Custom HolmesGPT Flags:**
- `--generate-mocks` - Generate mock files during test execution
- `--regenerate-all-mocks` - Regenerate all mock files (implies --generate-mocks)
- `--skip-setup` - Skip `before_test` commands
- `--skip-cleanup` - Skip `after_test` commands

**Common Pytest Flags:**
- `-n <number>` - Run tests in parallel
- `-k <pattern>` - Run tests matching pattern
- `-m <marker>` - Run tests with specific marker
- `-v/-vv` - Verbose output
- `-s` - Show print statements
- `--no-cov` - Disable coverage
- `--collect-only` - List tests without running

This completes the evaluation writing guide. The next step is setting up reporting and analysis using Braintrust.
6 changes: 4 additions & 2 deletions holmes/core/tool_calling_llm.py
Original file line number Diff line number Diff line change
Expand Up @@ -39,7 +39,7 @@
)
from holmes.utils.tags import format_tags_in_string, parse_messages_tags
from holmes.core.tools_utils.tool_executor import ToolExecutor
from holmes.core.tracing import DummySpan, SpanType
from holmes.core.tracing import DummySpan


def format_tool_result_data(tool_result: StructuredToolResult) -> str:
Expand Down Expand Up @@ -422,7 +422,7 @@ def _invoke_tool(
tool_response = None

# Create tool span if tracing is enabled
tool_span = trace_span.start_span(name=tool_name, type=SpanType.TOOL)
tool_span = trace_span.start_span(name=tool_name, type="tool")

try:
tool_response = prevent_overly_repeated_tool_call(
Expand Down Expand Up @@ -451,6 +451,8 @@ def _invoke_tool(
metadata={
"status": tool_response.status.value,
"error": tool_response.error,
"description": tool.get_parameterized_one_liner(tool_params),
"structured_tool_result": tool_response,
},
)

Expand Down
9 changes: 5 additions & 4 deletions holmes/core/tracing.py
Original file line number Diff line number Diff line change
Expand Up @@ -32,12 +32,13 @@ class SpanType(Enum):
TOOL = "tool"
TASK = "task"
SCORE = "score"
EVAL = "eval"


class DummySpan:
"""A no-op span implementation for when tracing is disabled."""

def start_span(self, name: str, span_type: Optional[SpanType] = None, **kwargs):
def start_span(self, name: str, span_type=None, **kwargs):
return DummySpan()

def log(self, *args, **kwargs):
Expand Down Expand Up @@ -121,17 +122,17 @@ def start_trace(
# Add span type to kwargs if provided
kwargs = {}
if span_type:
kwargs["type"] = getattr(SpanTypeAttribute, span_type.name)
kwargs["type"] = span_type.value

# Use current Braintrust context (experiment or parent span)
current_span = braintrust.current_span()
if not _is_noop_span(current_span):
return current_span.start_span(name=name, **kwargs)
return current_span.start_span(name=name, **kwargs) # type: ignore

# Fallback to current experiment
current_experiment = braintrust.current_experiment()
if current_experiment:
return current_experiment.start_span(name=name, **kwargs)
return current_experiment.start_span(name=name, **kwargs) # type: ignore

return DummySpan()

Expand Down
Loading