Skip to content
Merged
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
50 changes: 49 additions & 1 deletion tests/llm/utils/reporting/github_reporter.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,7 @@
import logging
import os
from typing import Dict, List, Optional, Tuple
from urllib.parse import quote

from tests.llm.utils.braintrust import get_braintrust_url
from tests.llm.utils.braintrust_history import (
Expand All @@ -14,9 +15,44 @@
compare_with_benchmark,
get_benchmark_baseline,
)
from tests.llm.utils.test_env_vars import GITHUB_REF_NAME
from tests.llm.utils.test_results import TestStatus


_TEST_TYPE_TO_FIXTURE_DIR = {
"ask": "test_ask_holmes",
"investigate": "test_investigate",
}


def _get_eval_source_url(test_type: str, test_case_name: str) -> Optional[str]:
"""Build a GitHub URL to an eval's test_case.yaml on the branch this run executed from.

Returns None if the test_type is not a known fixture-backed test (e.g. "unknown").

Ref resolution (first non-empty wins):
1. EVAL_BRANCH — explicit override
2. GITHUB_HEAD_REF — PR head branch (only set on pull_request events;
GITHUB_REF_NAME on PRs is the virtual "<num>/merge" ref which is not browsable)
3. GITHUB_REF_NAME — branch name on push events
4. "master"
"""
fixture_dir = _TEST_TYPE_TO_FIXTURE_DIR.get(test_type)
if not fixture_dir or not test_case_name:
return None
ref = (
os.environ.get("EVAL_BRANCH")
or os.environ.get("GITHUB_HEAD_REF")
or GITHUB_REF_NAME
or "master"
)
encoded_ref = quote(ref, safe="")
return (
f"https://github.com/HolmesGPT/holmesgpt/blob/{encoded_ref}"
f"/tests/llm/fixtures/{fixture_dir}/{test_case_name}/test_case.yaml"
)
Comment thread
aantn marked this conversation as resolved.


def _fmt_tokens(value: Optional[int]) -> str:
"""Format a token count: comma-separated if present, dash if absent/zero."""
if value is not None and value > 0:
Expand Down Expand Up @@ -63,8 +99,13 @@ def _generate_comparison_tables(
comparison = comparison_map.get(key)
baseline = benchmark.get(key)

display_name = f"{test_name} ({model})" if model else test_name
source_url = _get_eval_source_url(result.get("test_type", ""), test_name)
if source_url:
display_name = f"{display_name} [📄]({source_url})"

rows.append({
"name": f"{test_name} ({model})" if model else test_name,
"name": display_name,
"current_time": result.get("holmes_duration"),
"baseline_time": baseline.duration if baseline else None,
"current_cost": result.get("cost"),
Expand Down Expand Up @@ -341,6 +382,13 @@ def generate_markdown_report(
if braintrust_url:
test_case_name = f"[{test_case_name}]({braintrust_url})"

# Add 📄 link to the eval's test_case.yaml on the branch this run ran from
source_url = _get_eval_source_url(
result.get("test_type", ""), result["test_case_name"]
)
if source_url:
test_case_name = f"[📄]({source_url}) {test_case_name}"

status = TestStatus(result)

# Format time (plain, no inline comparison)
Expand Down
Loading