diff --git a/.github/workflows/llm-evaluation.yaml b/.github/workflows/llm-evaluation.yaml index d65111011e..48bee29da7 100644 --- a/.github/workflows/llm-evaluation.yaml +++ b/.github/workflows/llm-evaluation.yaml @@ -43,7 +43,6 @@ jobs: OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }} BRAINTRUST_API_KEY: ${{ secrets.BRAINTRUST_API_KEY }} UPLOAD_DATASET: "true" - PUSH_EVALS_TO_BRAINTRUST: "true" EXPERIMENT_ID: github-${{ github.run_id }}.${{ github.run_number }}.${{ github.run_attempt }} run: | poetry run pytest --no-cov tests/llm/test_ask_holmes.py tests/llm/test_investigate.py -n 6 diff --git a/CLAUDE.md b/CLAUDE.md index eea981eadd..338bc8d4bc 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -137,7 +137,8 @@ poetry run pytest tests/llm/test_ask_holmes.py - `OPENAI_API_KEY`, `ANTHROPIC_API_KEY`: LLM API keys - `MODEL`: Override default model - `RUN_LIVE`: Use live tools in tests instead of mocks -- `BRAINTRUST_API_KEY`: For test result tracking +- `BRAINTRUST_API_KEY`: For test result tracking and CI/CD report generation +- `BRAINTRUST_ORG`: Braintrust organization name (default: "robustadev") ## Development Guidelines diff --git a/docs/development/evals/index.md b/docs/development/evals/index.md index 9135697108..69d79bcb49 100644 --- a/docs/development/evals/index.md +++ b/docs/development/evals/index.md @@ -100,6 +100,8 @@ Configure evaluations using these environment variables: | `RUN_LIVE` | `RUN_LIVE=true` | Execute `before-test` and `after-test` commands, ignore mock files | | `UPLOAD_DATASET` | `UPLOAD_DATASET=true` | Sync dataset to external evaluation platform | | `EXPERIMENT_ID` | `EXPERIMENT_ID=my_baseline` | Custom experiment name for result tracking | +| `BRAINTRUST_API_KEY` | `BRAINTRUST_API_KEY=sk-...` | Enable Braintrust integration for result tracking and CI/CD report generation | +| `BRAINTRUST_ORG` | `BRAINTRUST_ORG=my-org` | Braintrust organization name (defaults to "robustadev") | ### Simple Example diff --git a/docs/development/evals/reporting.md b/docs/development/evals/reporting.md index 77710e064f..e4b9ea5740 100644 --- a/docs/development/evals/reporting.md +++ b/docs/development/evals/reporting.md @@ -40,7 +40,6 @@ export BRAINTRUST_API_KEY=sk-your-api-key-here ```bash export BRAINTRUST_API_KEY=sk-your-key export UPLOAD_DATASET=true -export PUSH_EVALS_TO_BRAINTRUST=true pytest ./tests/llm/test_ask_holmes.py ``` @@ -58,7 +57,6 @@ pytest -n 10 ./tests/llm/test_*.py | Variable | Purpose | |----------|---------| | `UPLOAD_DATASET` | Sync test cases to Braintrust | -| `PUSH_EVALS_TO_BRAINTRUST` | Upload evaluation results | | `EXPERIMENT_ID` | Name your experiment run. This makes it easier to find and track in Braintrust's UI | | `MODEL` | The LLM model for Holmes to use | | `CLASSIFIER_MODEL` | The LLM model to use for scoring the answer (LLM as judge) | diff --git a/tests/llm/conftest.py b/tests/llm/conftest.py index 7e96187fa5..14eabd09e3 100644 --- a/tests/llm/conftest.py +++ b/tests/llm/conftest.py @@ -1,12 +1,47 @@ +# Standard library imports import logging import os -import pytest +import textwrap from contextlib import contextmanager -from tests.llm.utils.braintrust import get_experiment_results, get_experiment_name -from braintrust.span_types import SpanTypeAttribute +from dataclasses import dataclass +from typing import List, Optional + +# Third-party imports +import pytest +from litellm import completion +from rich.console import Console +from rich.table import Table + +# Local imports +from tests.llm.utils.braintrust import get_experiment_name from tests.llm.utils.constants import PROJECT from tests.llm.utils.classifiers import create_llm_client +# Configuration constants +DEBUG_SEPARATOR = "=" * 80 + + +@dataclass +class TestResult: + test_id: str + test_name: str + expected: str + actual: str + pass_fail: str + tools_called: List[str] + logs: str + test_type: str = "" + error_message: Optional[str] = None + execution_time: Optional[float] = None + expected_correctness_score: float = 1.0 + actual_correctness_score: float = 0.0 + + +def pytest_configure(config): + """Configure pytest settings""" + # Suppress noisy LiteLLM logs during testing + logging.getLogger("LiteLLM").setLevel(logging.WARNING) + @contextmanager def force_pytest_output(request): @@ -148,14 +183,34 @@ def braintrust_eval_link(request): test_case_id = request.node.name # Construct Braintrust URL for this specific test - braintrust_url = f"https://www.braintrust.dev/app/robustadev/p/{PROJECT}/experiments/{experiment_name}?r=&s=&c={test_case_id}" + braintrust_org = os.environ.get("BRAINTRUST_ORG", "robustadev") + braintrust_url = f"https://www.braintrust.dev/app/{braintrust_org}/p/{PROJECT}/experiments/{experiment_name}?r=&s=&c={test_case_id}" with force_pytest_output(request): print(f"\nšŸ” View eval result: {braintrust_url}") print() +def pytest_terminal_summary(terminalreporter, exitstatus, config): + """Generate GitHub Actions report and Rich summary table from terminalreporter.stats (xdist compatible)""" + if not hasattr(terminalreporter, "stats"): + return + + # Collect and sort test results from terminalreporter.stats + sorted_results = _collect_test_results_from_stats(terminalreporter) + + if not sorted_results: + return + + # Handle GitHub/CI output (markdown + file writing) + _handle_github_output(sorted_results) + + # Handle console/developer output (Rich table + Braintrust links) + _handle_console_output(sorted_results) + + def markdown_table(headers, rows): + """Generate a markdown table from headers and rows.""" markdown = "| " + " | ".join(headers) + " |\n" markdown += "| " + " | ".join(["---" for _ in headers]) + " |\n" for row in rows: @@ -163,90 +218,368 @@ def markdown_table(headers, rows): return markdown -@pytest.mark.llm -def pytest_terminal_summary(terminalreporter, exitstatus, config): - if not os.environ.get("PUSH_EVALS_TO_BRAINTRUST"): - # The code fetches the evals from Braintrust to print out a summary. - # Skip running it if the evals have not been uploaded to Braintrust - return +def _collect_test_results_from_stats(terminalreporter): + """Collect and parse test results from terminalreporter.stats.""" + test_results = {} + + for status, reports in terminalreporter.stats.items(): + for report in reports: + # Only process 'call' phase reports for actual test results + if getattr(report, "when", None) != "call": + continue + + # Only process LLM evaluation tests + nodeid = getattr(report, "nodeid", "") + if not ("test_ask_holmes" in nodeid or "test_investigate" in nodeid): + continue + + # Extract test data from user_properties + user_props = dict(getattr(report, "user_properties", {})) + if not user_props: # Skip if no user_properties + continue + + # Extract test info + test_id = _extract_test_id_from_nodeid(nodeid) + test_name = _extract_test_name_from_nodeid(nodeid) + test_type = "ask" if "test_ask_holmes" in nodeid else "investigate" + + # Get test data from user_properties + expected = user_props.get("expected", "Unknown") + actual = user_props.get("actual", "Unknown") + tools_called = user_props.get("tools_called", []) + expected_correctness_score = float( + user_props.get("expected_correctness_score", 1.0) + ) + actual_correctness_score = float( + user_props.get("actual_correctness_score", 0.0) + ) - headers = ["Test suite", "Test case", "Status"] - rows = [] - - # Do not change the title below without updating the github workflow that references it - markdown = "## Results of HolmesGPT evals\n" - - for test_suite in ["ask_holmes", "investigate"]: - try: - result = get_experiment_results(PROJECT, test_suite) - result.records.sort(key=lambda x: x.get("span_attributes", {}).get("name")) - total_test_cases = 0 - successful_test_cases = 0 - regressions = 0 - for record in result.records: - scores = record.get("scores", None) - span_id = record.get("id") - span_attributes = record.get("span_attributes") - if scores and span_attributes: - span_type = span_attributes.get("type") - if span_type != SpanTypeAttribute.EVAL: - continue - - span_name = span_attributes.get("name") - test_case = next( - (tc for tc in result.test_cases if tc.get("id") == span_name), - {}, - ) - correctness_score = int(scores.get("correctness", 0)) - expected_correctness_score = ( - test_case.get("metadata", {}) - .get("test_case", {}) - .get("evaluation", {}) - .get("correctness", 0) - ) - if isinstance(expected_correctness_score, dict): - expected_correctness_score = expected_correctness_score.get( - "expected_score", 1 - ) - expected_correctness_score = int(expected_correctness_score) - - total_test_cases += 1 - if correctness_score == 1: - successful_test_cases += 1 - status_text = ":white_check_mark:" - elif correctness_score == 0 and expected_correctness_score == 0: - status_text = ":warning:" - else: - regressions += 1 - status_text = ":x:" - rows.append( - [ - f"[{test_suite}](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/{result.experiment_name})", - f"[{span_name}](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/{result.experiment_name}?r={span_id})", - status_text, - ] - ) - markdown += f"\n- [{test_suite}](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/{result.experiment_name}): {successful_test_cases}/{total_test_cases} test cases were successful, {regressions} regressions\n" + # Store result (use nodeid as key to avoid duplicates) + test_results[nodeid] = { + "test_id": test_id, + "test_name": test_name, + "test_type": test_type, + "expected": expected, + "actual": actual, + "tools_called": tools_called, + "expected_correctness_score": expected_correctness_score, + "actual_correctness_score": actual_correctness_score, + "status": status, # passed, failed, error, etc. + "outcome": getattr(report, "outcome", "unknown"), + "execution_time": getattr(report, "duration", None), + } + + # Sort results by test_type then test_id for consistent ordering + sorted_results = sorted( + test_results.values(), + key=lambda r: ( + r["test_type"], + int(r["test_id"]) if r["test_id"].isdigit() else 999, + r["test_name"], + ), + ) + + return sorted_results + + +def _get_braintrust_url(result): + """Generate Braintrust URL for a test result. + + Args: + result: Test result dictionary + + Returns: + Braintrust URL string, or None if Braintrust is not configured + """ + braintrust_api_key = os.environ.get("BRAINTRUST_API_KEY") + if not braintrust_api_key: + return None - except ValueError: - logging.info( - f"Failed to fetch braintrust experiment {PROJECT}-{test_suite}" - ) + test_suite = "ask_holmes" if result["test_type"] == "ask" else "investigate" + experiment_name = get_experiment_name(test_suite) + test_case_id = f"test_{test_suite}[{result['test_id']}_{result['test_name']}]" + + braintrust_org = os.environ.get("BRAINTRUST_ORG", "robustadev") + return ( + f"https://www.braintrust.dev/app/{braintrust_org}/p/{PROJECT}/" + f"experiments/{experiment_name}?r=&s=&c={test_case_id}" + ) + + +def _generate_markdown_report(sorted_results): + """Generate markdown report from sorted test results.""" + markdown = "## Results of HolmesGPT evals\n\n" + + # Count results by test type and status using proper regression logic + ask_holmes_total = ask_holmes_passed = ask_holmes_regressions = 0 + investigate_total = investigate_passed = investigate_regressions = 0 + + for result in sorted_results: + actual_score = int(result["actual_correctness_score"]) + expected_score = int(result["expected_correctness_score"]) + + if result["test_type"] == "ask": + ask_holmes_total += 1 + if actual_score == 1: + ask_holmes_passed += 1 + elif actual_score == 0 and expected_score == 0: + # Known failure, not a regression + pass + else: + ask_holmes_regressions += 1 + elif result["test_type"] == "investigate": + investigate_total += 1 + if actual_score == 1: + investigate_passed += 1 + elif actual_score == 0 and expected_score == 0: + # Known failure, not a regression + pass + else: + investigate_regressions += 1 + + # Generate summary lines + if ask_holmes_total > 0: + markdown += f"- ask_holmes: {ask_holmes_passed}/{ask_holmes_total} test cases were successful, {ask_holmes_regressions} regressions\n" + if investigate_total > 0: + markdown += f"- investigate: {investigate_passed}/{investigate_total} test cases were successful, {investigate_regressions} regressions\n" + + # Generate detailed table + markdown += "\n\n| Test suite | Test case | Status |\n" + markdown += "| --- | --- | --- |\n" + + for result in sorted_results: + test_suite = result["test_type"] + test_name = f"{result['test_id']}: {result['test_name']}" + + # Add Braintrust link to test name if available + braintrust_url = _get_braintrust_url(result) + if braintrust_url: + test_name = f"[{test_name}]({braintrust_url})" + + actual_score = int(result["actual_correctness_score"]) + expected_score = int(result["expected_correctness_score"]) + + if actual_score == 1: + status = ":white_check_mark:" + elif actual_score == 0 and expected_score == 0: + status = ":warning:" # Known failure + else: + status = ":x:" # Regression - if len(rows) > 0: - markdown += "\n\n" - markdown += markdown_table(headers, rows) - markdown += "\n\n**Legend**\n" - markdown += "\n- :white_check_mark: the test was successful" - markdown += ( - "\n- :warning: the test failed but is known to be flakky or known to fail" - ) - markdown += "\n- :x: the test failed and should be fixed before merging the PR" + markdown += f"| {test_suite} | {test_name} | {status} |\n" + + markdown += "\n\n**Legend**\n" + markdown += "\n- :white_check_mark: the test was successful" + markdown += ( + "\n- :warning: the test failed but is known to be flaky or known to fail" + ) + markdown += "\n- :x: the test failed and should be fixed before merging the PR" + + return markdown, sorted_results, ask_holmes_regressions + investigate_regressions + + +def _handle_github_output(sorted_results): + """Generate and write GitHub Actions report files.""" + # Generate markdown report + markdown, _, total_regressions = _generate_markdown_report(sorted_results) + # Write report files if Braintrust is configured + braintrust_api_key = os.environ.get("BRAINTRUST_API_KEY") + if braintrust_api_key: with open("evals_report.txt", "w", encoding="utf-8") as file: file.write(markdown) - # write number of regresssion to a separate file, if there are any regressions - if regressions > 0: + # Write regressions file if needed + if total_regressions > 0: with open("regressions.txt", "w", encoding="utf-8") as file: - file.write(f"{regressions}") + file.write(f"{total_regressions}") + + +def _extract_test_id_from_nodeid(nodeid: str) -> str: + """Extract test ID from pytest nodeid. + + Args: + nodeid: Pytest node ID like 'test_ask_holmes[01_how_many_pods]' + + Returns: + Test ID like '01', or 'unknown' if not found + """ + if "[" in nodeid and "]" in nodeid: + test_case = nodeid.split("[")[1].split("]")[0] + # Extract number from start of test case name + return test_case.split("_")[0] if "_" in test_case else test_case + return "unknown" + + +def _extract_test_name_from_nodeid(nodeid: str) -> str: + """Extract readable test name from pytest nodeid. + + Args: + nodeid: Pytest node ID like 'test_ask_holmes[01_how_many_pods]' + + Returns: + Test name like 'how_many_pods' + """ + try: + if "[" in nodeid and "]" in nodeid: + test_case = nodeid.split("[")[1].split("]")[0] + # Remove number prefix and convert underscores to spaces + parts = test_case.split("_")[1:] if "_" in test_case else [test_case] + return "_".join(parts) + except (IndexError, AttributeError): + pass + return nodeid.split("::")[-1] if "::" in nodeid else nodeid + + +def _handle_console_output(sorted_results): + """Display Rich table and Braintrust links for developers.""" + if not sorted_results: + return + + # Create Rich table + console = Console() + table = Table( + title="šŸ” HOLMES TESTS SUMMARY", + show_header=True, + header_style="bold magenta", + show_lines=True, + ) + + # Add columns with specific widths + table.add_column("Test", style="cyan", width=25) + table.add_column("Status", justify="center", width=4) + table.add_column("Time", justify="right", width=6) + table.add_column("Expected", style="green", width=35) + table.add_column("Actual", style="yellow", width=35) + table.add_column("Analysis", style="red", width=40) + + # Add rows to table + for result in sorted_results: + # Determine pass/fail status + actual_score = int(result["actual_correctness_score"]) + pass_fail = "āœ… PASS" if actual_score == 1 else "āŒ FAIL" + + # Create TestResult object for analysis function + test_result = TestResult( + test_id=result["test_id"], + test_name=result["test_name"], + expected=result["expected"], + actual=result["actual"], + pass_fail=pass_fail, + tools_called=result["tools_called"], + logs="", # We don't have logs in this context + test_type=result["test_type"], + error_message=None, + execution_time=result.get("execution_time"), + expected_correctness_score=result["expected_correctness_score"], + actual_correctness_score=result["actual_correctness_score"], + ) + + # Wrap long content for table readability + expected_wrapped = ( + "\n".join(textwrap.wrap(result["expected"], width=33)) + if result["expected"] + else "" + ) + actual_wrapped = ( + "\n".join(textwrap.wrap(result["actual"], width=33)) + if result["actual"] + else "" + ) + + # Combine test ID and name + combined_test_name = ( + f"{result['test_id']}_{result['test_name']} ({result['test_type']})" + ) + # Wrap test name to fit column + test_name_wrapped = "\n".join(textwrap.wrap(combined_test_name, width=23)) + + # Convert pass/fail to check/x status with colors + if "PASS" in pass_fail: + status = "[green]āœ“[/green]" + else: + status = "[red]āœ—[/red]" + + # Format execution time + time_str = ( + f"{result.get('execution_time'):.1f}s" + if result.get("execution_time") + else "N/A" + ) + + # Get analysis for failed tests + analysis = _get_analysis_for_result(test_result) + + table.add_row( + test_name_wrapped, + status, + time_str, + expected_wrapped, + actual_wrapped, + analysis, + ) + + console.print(table) + + # Print Braintrust links if enabled + if os.environ.get("BRAINTRUST_API_KEY"): + print("šŸ” BRAINTRUST EVAL LINKS:") + for result in sorted_results: + braintrust_url = _get_braintrust_url(result) + if braintrust_url: + print( + f"* {result['test_id']}_{result['test_name']} " + f"({result['test_type']}) - {braintrust_url}" + ) + print(DEBUG_SEPARATOR) + + +def _get_analysis_for_result(result): + """Get analysis text for a test result, with proper text wrapping.""" + if "PASS" in result.pass_fail: + return "" + + try: + analysis = _get_llm_analysis(result) + # Wrap analysis text for table readability + return "\n".join(textwrap.wrap(analysis, width=38)) + except Exception as e: + return f"Analysis failed: {str(e)}" + + +def _get_llm_analysis(result: TestResult) -> str: + """Get LLM analysis of test failure using GPT-4o. + + Args: + result: TestResult object containing test details + + Returns: + Analysis text explaining why the test failed + """ + prompt = textwrap.dedent(f"""\ + Analyze this failed eval for an AIOps agent why it failed. + TEST: {result.test_name} + EXPECTED: {result.expected} + ACTUAL: {result.actual} + TOOLS CALLED: {', '.join(result.tools_called)} + ERROR: {result.error_message or 'Test assertion failed'} + + LOGS: + {result.logs if result.logs else 'No logs available'} + + Please provide a concise analysis (2-3 sentences) and categorize this as one of: + - Problem with mock data - the test is failing due to incorrect or incomplete mock data, but the agent itself did the correct queries you would expect it to do + - Setup issue - the test is failing due to an issue with the test setup, such as missing tools or incorrect before_test/after_test configuration + - Real failure - the test is failing because the agent did not perform as expected, and this is a real issue that needs to be fixed + """) + + try: + response = completion( + model="gpt-4o", + messages=[{"role": "user", "content": prompt}], + max_tokens=200, + temperature=0.1, + ) + return response.choices[0].message.content.strip() + except Exception as e: + return f"Analysis failed: {e}" diff --git a/tests/llm/test_ask_holmes.py b/tests/llm/test_ask_holmes.py index ca77ebb16a..75d7210cad 100644 --- a/tests/llm/test_ask_holmes.py +++ b/tests/llm/test_ask_holmes.py @@ -62,7 +62,9 @@ def idfn(val): @pytest.mark.llm @pytest.mark.parametrize("experiment_name, test_case", get_test_cases(), ids=idfn) -def test_ask_holmes(experiment_name: str, test_case: AskHolmesTestCase, caplog): +def test_ask_holmes( + experiment_name: str, test_case: AskHolmesTestCase, caplog, request +): tracer = TracingFactory.create_tracer("braintrust", project=PROJECT) # Create experiment using unified API @@ -172,6 +174,27 @@ def test_ask_holmes(experiment_name: str, test_case: AskHolmesTestCase, caplog): print(f"\n** OUTPUT **\n{output}") print(f"\n** SCORES **\n{scores}") + # Store data for summary plugin + expected_correctness_score = ( + test_case.evaluation.correctness.expected_score + if isinstance(test_case.evaluation.correctness, Evaluation) + else test_case.evaluation.correctness + ) + request.node.user_properties.append(("expected", debug_expected)) + request.node.user_properties.append(("actual", output or "")) + request.node.user_properties.append( + ( + "tools_called", + tools_called if isinstance(tools_called, list) else [str(tools_called)], + ) + ) + request.node.user_properties.append( + ("expected_correctness_score", expected_correctness_score) + ) + request.node.user_properties.append( + ("actual_correctness_score", scores.get("correctness", 0)) + ) + assert ( int(scores.get("correctness", 0)) == 1 ), f"Test {test_case.id} failed (score: {scores.get('correctness', 0)})\nActual: {output}\nExpected: {debug_expected}" diff --git a/tests/llm/test_investigate.py b/tests/llm/test_investigate.py index 5ed3a7e2bd..160d309755 100644 --- a/tests/llm/test_investigate.py +++ b/tests/llm/test_investigate.py @@ -21,7 +21,7 @@ from tests.llm.utils.system import get_machine_state_tags from tests.llm.utils.mock_dal import MockSupabaseDal from tests.llm.utils.mock_toolset import MockToolsets -from tests.llm.utils.mock_utils import InvestigateTestCase, MockHelper +from tests.llm.utils.mock_utils import InvestigateTestCase, MockHelper, Evaluation from os import path from unittest.mock import patch @@ -43,7 +43,6 @@ def create_tool_executor(self, dal: Optional[SupabaseDal]) -> ToolExecutor: mock = MockToolsets( generate_mocks=self._test_case.generate_mocks, test_case_folder=self._test_case.folder, - parent_span=None, # Use tracer context instead ) expected_tools = [] @@ -105,7 +104,9 @@ def idfn(val): @pytest.mark.llm @pytest.mark.parametrize("experiment_name, test_case", get_test_cases(), ids=idfn) -def test_investigate(experiment_name: str, test_case: InvestigateTestCase, caplog): +def test_investigate( + experiment_name: str, test_case: InvestigateTestCase, caplog, request +): # Use unified tracing API for evals from holmes.core.tracing import TracingFactory @@ -193,6 +194,27 @@ def test_investigate(experiment_name: str, test_case: InvestigateTestCase, caplo print(f"\n** OUTPUT **\n{output}") print(f"\n** SCORES **\n{scores}") + # Store data for summary plugin + expected_correctness_score = ( + test_case.evaluation.correctness.expected_score + if isinstance(test_case.evaluation.correctness, Evaluation) + else test_case.evaluation.correctness + ) + request.node.user_properties.append(("expected", debug_expected)) + request.node.user_properties.append(("actual", output or "")) + request.node.user_properties.append( + ( + "tools_called", + tools_called if isinstance(tools_called, list) else [str(tools_called)], + ) + ) + request.node.user_properties.append( + ("expected_correctness_score", expected_correctness_score) + ) + request.node.user_properties.append( + ("actual_correctness_score", scores.get("correctness", 0)) + ) + assert result.sections, "Missing sections" assert ( len(result.sections) >= len(investigate_request.sections) diff --git a/tests/llm/test_workload_health.py b/tests/llm/test_workload_health.py index 069fadb267..f30c25828e 100644 --- a/tests/llm/test_workload_health.py +++ b/tests/llm/test_workload_health.py @@ -95,7 +95,9 @@ def idfn(val): @pytest.mark.llm @pytest.mark.parametrize("experiment_name, test_case", get_test_cases(), ids=idfn) -def test_health_check(experiment_name: str, test_case: HealthCheckTestCase, caplog): +def test_health_check( + experiment_name: str, test_case: HealthCheckTestCase, caplog, request +): dataset_name = braintrust_util.get_dataset_name("health_check") bt_helper = braintrust_util.BraintrustEvalHelper( project_name=PROJECT, dataset_name=dataset_name @@ -158,6 +160,16 @@ def test_health_check(experiment_name: str, test_case: HealthCheckTestCase, capl print(f"\n** OUTPUT **\n{output}") print(f"\n** SCORES **\n{scores}") + # Store data for summary plugin + request.node.user_properties.append(("expected", debug_expected)) + request.node.user_properties.append(("actual", output or "")) + request.node.user_properties.append( + ( + "tools_called", + tools_called if isinstance(tools_called, list) else [str(tools_called)], + ) + ) + if test_case.evaluation.correctness: expected_correctness = test_case.evaluation.correctness if isinstance(expected_correctness, Evaluation):