From 21ab7cb994c04a5a0a1f2a9af07c8ef387360365 Mon Sep 17 00:00:00 2001 From: Tomer Keshet Date: Sun, 4 Jan 2026 18:28:16 +0200 Subject: [PATCH 1/7] WIP benchmarks fast Signed-off-by: Tomer Keshet --- .github/workflows/eval-benchmarks.yaml | 186 ++++++++++++++---- docs/development/evaluations/.nav.yml | 3 +- .../evaluations/fast-benchmark-results.md | 71 +++++++ .../evaluations/history/fast/.gitkeep | 0 .../evaluations/history/{ => fast}/.nav.yml | 0 .../evaluations/history/fast/index.md | 12 ++ .../history/fast/results_20260104_174301.md | 121 ++++++++++++ .../history/fast/results_20260104_174350.md | 22 +++ .../history/fast/results_20260104_174620.md | 70 +++++++ .../evaluations/history/full/.gitkeep | 0 .../evaluations/history/full/.nav.yml | 5 + .../custom_claude_results_20250930_153753.md | 0 ...tom_self_hosted_results_20251008_053744.md | 0 .../evaluations/history/full/index.md | 16 ++ .../{ => full}/results_20250928_001434.md | 0 .../{ => full}/results_20250930_085923.md | 0 .../{ => full}/results_20251012_170303.md | 0 .../{ => full}/results_20251127_042958.md | 0 docs/development/evaluations/history/index.md | 9 - docs/development/evaluations/index.md | 15 +- .../development/evaluations/latest-results.md | 100 +++------- run_benchmarks_local.py | 122 ++++++++++-- tests/generate_eval_report.py | 42 +++- 23 files changed, 646 insertions(+), 148 deletions(-) create mode 100644 docs/development/evaluations/fast-benchmark-results.md create mode 100644 docs/development/evaluations/history/fast/.gitkeep rename docs/development/evaluations/history/{ => fast}/.nav.yml (100%) create mode 100644 docs/development/evaluations/history/fast/index.md create mode 100644 docs/development/evaluations/history/fast/results_20260104_174301.md create mode 100644 docs/development/evaluations/history/fast/results_20260104_174350.md create mode 100644 docs/development/evaluations/history/fast/results_20260104_174620.md create mode 100644 docs/development/evaluations/history/full/.gitkeep create mode 100644 docs/development/evaluations/history/full/.nav.yml rename docs/development/evaluations/history/{ => full}/custom_claude_results_20250930_153753.md (100%) rename docs/development/evaluations/history/{ => full}/custom_self_hosted_results_20251008_053744.md (100%) create mode 100644 docs/development/evaluations/history/full/index.md rename docs/development/evaluations/history/{ => full}/results_20250928_001434.md (100%) rename docs/development/evaluations/history/{ => full}/results_20250930_085923.md (100%) rename docs/development/evaluations/history/{ => full}/results_20251012_170303.md (100%) rename docs/development/evaluations/history/{ => full}/results_20251127_042958.md (100%) delete mode 100644 docs/development/evaluations/history/index.md diff --git a/.github/workflows/eval-benchmarks.yaml b/.github/workflows/eval-benchmarks.yaml index 090101de84..6e7efd5042 100644 --- a/.github/workflows/eval-benchmarks.yaml +++ b/.github/workflows/eval-benchmarks.yaml @@ -1,27 +1,33 @@ -name: Run full eval benchmarks +name: Run eval benchmarks on: # Allow manual trigger workflow_dispatch: inputs: + benchmark_type: + description: 'Benchmark type (fast-benchmark or full-benchmark). Cannot be combined with custom test_markers.' + required: false + type: choice + options: + - 'fast-benchmark' + - 'full-benchmark' + default: 'fast-benchmark' models: - description: 'Comma-separated list of models to test (e.g., gpt-4o,claude-sonnet-4)' + description: 'Comma-separated list of models to test (leave empty for defaults from run_benchmarks_local.py)' required: false - default: 'gpt-4o,gpt-4.1,gpt-5,anthropic/claude-sonnet-4-20250514' + default: '' test_markers: - description: 'Additional pytest markers (will be combined with "llm" - e.g., "easy", "medium", "logs")' + description: 'Custom pytest markers (ONLY use if not using benchmark_type). Cannot be combined with benchmark_type.' required: false - default: 'easy' + default: '' iterations: description: 'Number of iterations per test (max 10)' required: false - # TODO: For testing, use just 1 iteration by default - default: '1' # Was: '3' + default: '1' - # TODO: Enable after testing - # # Run weekly on Sunday at 2 AM UTC - # schedule: - # - cron: '0 2 * * 0' + # Run weekly on Sunday at 2 AM UTC + schedule: + - cron: '0 2 * * 0' jobs: run-benchmarks: @@ -51,24 +57,56 @@ jobs: - name: Determine test command id: test-command run: | - # Set default values from workflow inputs or triggers + # Define benchmark type markers + FAST_MARKERS="regression or benchmark" + FULL_MARKERS="easy or medium or hard or regression or benchmark" + + # Default models from run_benchmarks_local.py + DEFAULT_MODELS="gpt-5.1,gpt-5,sonnet-4.5,haiku-4.5,deepseek-3.1" + if [ "${{ github.event_name }}" == "workflow_dispatch" ]; then + BENCHMARK_TYPE="${{ github.event.inputs.benchmark_type }}" + CUSTOM_MARKERS="${{ github.event.inputs.test_markers }}" MODEL="${{ github.event.inputs.models }}" - # Always prepend "llm and " to user-provided markers with proper parentheses - TEST_MARKERS="llm and (${{ github.event.inputs.test_markers }})" - # Cap iterations at 10 ITERATIONS="${{ github.event.inputs.iterations }}" + + # Validate: cannot combine benchmark_type with custom markers + if [ -n "$CUSTOM_MARKERS" ] && [ -n "$BENCHMARK_TYPE" ]; then + echo "ERROR: Cannot combine benchmark_type with custom test_markers" + exit 1 + fi + + # Determine markers based on benchmark type or custom markers + if [ -n "$CUSTOM_MARKERS" ]; then + TEST_MARKERS="llm and ($CUSTOM_MARKERS)" + BENCHMARK_TYPE="" + elif [ "$BENCHMARK_TYPE" == "full-benchmark" ]; then + TEST_MARKERS="llm and ($FULL_MARKERS)" + else + # Default to fast-benchmark + BENCHMARK_TYPE="fast-benchmark" + TEST_MARKERS="llm and ($FAST_MARKERS)" + fi + + # Use default models if not specified + if [ -z "$MODEL" ]; then + MODEL="$DEFAULT_MODELS" + fi + + # Cap iterations at 10 if [ "$ITERATIONS" -gt 10 ]; then echo "Capping iterations at 10 (requested: $ITERATIONS)" ITERATIONS="10" fi elif [ "${{ github.event_name }}" == "schedule" ]; then - MODEL="gpt-4o,anthropic/claude-sonnet-4-20250514,gpt-4.1" - TEST_MARKERS="llm and (easy)" - ITERATIONS="10" + # Weekly scheduled run: fast-benchmark with defaults + BENCHMARK_TYPE="fast-benchmark" + MODEL="$DEFAULT_MODELS" + TEST_MARKERS="llm and ($FAST_MARKERS)" + ITERATIONS="1" fi - # Set test path and markers separately + # Set test path TEST_PATH="tests/llm/" # Write all outputs atomically @@ -77,6 +115,7 @@ jobs: echo "test_path=$TEST_PATH" echo "test_markers=$TEST_MARKERS" echo "iterations=$ITERATIONS" + echo "benchmark_type=$BENCHMARK_TYPE" } >> $GITHUB_OUTPUT - name: Run evaluation benchmarks @@ -115,18 +154,49 @@ jobs: - name: Generate benchmark report if: always() run: | - # Generate latest results + BENCHMARK_TYPE="${{ steps.test-command.outputs.benchmark_type }}" + TIMESTAMP=$(date +%Y%m%d_%H%M%S) + + # Determine output paths based on benchmark type + if [ "$BENCHMARK_TYPE" == "fast-benchmark" ]; then + MAIN_OUTPUT="docs/development/evaluations/fast-benchmark-results.md" + HISTORY_DIR="docs/development/evaluations/history/fast" + elif [ "$BENCHMARK_TYPE" == "full-benchmark" ]; then + MAIN_OUTPUT="docs/development/evaluations/full-benchmark-results.md" + HISTORY_DIR="docs/development/evaluations/history/full" + else + # Custom markers - use full benchmark location + MAIN_OUTPUT="docs/development/evaluations/full-benchmark-results.md" + HISTORY_DIR="docs/development/evaluations/history/full" + fi + + # Create history directory if needed + mkdir -p "$HISTORY_DIR" + + # Build benchmark-type argument if specified + BENCHMARK_ARG="" + if [ -n "$BENCHMARK_TYPE" ]; then + BENCHMARK_ARG="--benchmark-type $BENCHMARK_TYPE" + fi + + # Generate main results poetry run python tests/generate_eval_report.py \ --json-file eval_results.json \ - --output-file docs/development/evaluations/latest-results.md \ - --models "${{ steps.test-command.outputs.models }}" + --output-file "$MAIN_OUTPUT" \ + --models "${{ steps.test-command.outputs.models }}" \ + $BENCHMARK_ARG - # Also generate timestamped version for history - TIMESTAMP=$(date +%Y%m%d_%H%M%S) + # For fast-benchmark, also copy to latest-results.md for backwards compatibility + if [ "$BENCHMARK_TYPE" == "fast-benchmark" ]; then + cp "$MAIN_OUTPUT" docs/development/evaluations/latest-results.md + fi + + # Generate timestamped version for history poetry run python tests/generate_eval_report.py \ --json-file eval_results.json \ - --output-file "docs/development/evaluations/history/weekly/results_${TIMESTAMP}.md" \ - --models "${{ steps.test-command.outputs.models }}" + --output-file "${HISTORY_DIR}/results_${TIMESTAMP}.md" \ + --models "${{ steps.test-command.outputs.models }}" \ + $BENCHMARK_ARG - name: Upload eval results if: always() @@ -134,20 +204,54 @@ jobs: with: name: eval-results-${{ github.run_id }} path: | + docs/development/evaluations/fast-benchmark-results.md + docs/development/evaluations/full-benchmark-results.md docs/development/evaluations/latest-results.md - # TODO: Enable after testing - # - name: Commit benchmark results - # if: (github.event_name == 'schedule' || github.event_name == 'workflow_dispatch') && github.ref == 'refs/heads/main' - # run: | - # git config --local user.email "github-actions[bot]@users.noreply.github.com" - # git config --local user.name "github-actions[bot]" - # - # # Copy results to historical directory with timestamp - # TIMESTAMP=$(date +%Y%m%d_%H%M%S) - # mkdir -p docs/development/evaluations/history - # cp docs/development/evaluations/latest-results.md "docs/development/evaluations/history/results_${TIMESTAMP}.md" - # - # git add docs/development/evaluations/ - # git diff --staged --quiet || git commit -m "Update benchmark results [skip ci]" - # git push + - name: Create PR with benchmark results + if: always() && (github.event_name == 'schedule' || github.event_name == 'workflow_dispatch') + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + run: | + # Configure git + git config --local user.email "github-actions[bot]@users.noreply.github.com" + git config --local user.name "github-actions[bot]" + + # Create timestamped branch + BRANCH_NAME="automated/benchmark-$(date +%Y%m%d)" + git checkout -b "$BRANCH_NAME" + + # Add all evaluation results + git add docs/development/evaluations/ + + # Check if there are changes to commit + if git diff --staged --quiet; then + echo "No changes to commit" + exit 0 + fi + + # Commit and push + BENCHMARK_TYPE="${{ steps.test-command.outputs.benchmark_type }}" + git commit -m "Update ${BENCHMARK_TYPE:-benchmark} results [skip ci]" + git push origin "$BRANCH_NAME" --force + + # Create PR (or update existing) + PR_TITLE="Weekly Benchmark Results $(date +%Y-%m-%d)" + PR_BODY="Automated weekly benchmark results from CI. + + **Benchmark Type**: ${BENCHMARK_TYPE:-custom} + **Models**: ${{ steps.test-command.outputs.models }} + **Iterations**: ${{ steps.test-command.outputs.iterations }} + **Markers**: ${{ steps.test-command.outputs.test_markers }}" + + # Check if PR already exists + EXISTING_PR=$(gh pr list --head "$BRANCH_NAME" --json number --jq '.[0].number') + if [ -n "$EXISTING_PR" ]; then + echo "Updating existing PR #$EXISTING_PR" + else + gh pr create \ + --title "$PR_TITLE" \ + --body "$PR_BODY" \ + --base master \ + --head "$BRANCH_NAME" + fi diff --git a/docs/development/evaluations/.nav.yml b/docs/development/evaluations/.nav.yml index 2e74d6550b..aeaeaa7145 100644 --- a/docs/development/evaluations/.nav.yml +++ b/docs/development/evaluations/.nav.yml @@ -1,7 +1,8 @@ nav: - index.md - Latest Results: latest-results.md - - Historical Results: history + - Fast Benchmark: history/fast + - Full Benchmark: history/full - Running Evaluations: running-evals.md - Adding New Evaluations: adding-evals.md - Benchmarking New Models: benchmarking-new-models.md diff --git a/docs/development/evaluations/fast-benchmark-results.md b/docs/development/evaluations/fast-benchmark-results.md new file mode 100644 index 0000000000..39f863ff4a --- /dev/null +++ b/docs/development/evaluations/fast-benchmark-results.md @@ -0,0 +1,71 @@ +# HolmesGPT LLM Evaluation Fast Benchmark Results + +**Generated**: 2026-01-04 18:10 UTC +**Total Duration**: 1m 18s +**Iterations**: 1 +**Judge (classifier) model**: gpt-4.1 + +!!! info "Fast Benchmark" + **Markers**: `regression or benchmark`
+ **Schedule**: Weekly (Sunday 2 AM UTC)
+ **Purpose**: Quick regression tests to catch breaking changes + +HolmesGPT is continuously evaluated against real-world Kubernetes and cloud troubleshooting scenarios. + +If you find scenarios that HolmesGPT does not perform well on, please consider adding them as evals to the benchmark. + +## Model Accuracy Comparison + +| Model | Pass | Fail | Skip/Error | Total | Success Rate | +|-------|------|------|------------|-------|--------------| +| sonnet-4.5 | 1 | 0 | 0 | 1 | 🟢 100% (1/1) | + +## Model Cost Comparison + +| Model | Tests | Avg Cost | Min Cost | Max Cost | Total Cost | +|-------|-------|----------|----------|----------|------------| +| sonnet-4.5 | 1 | $0.18 | $0.18 | $0.18 | $0.18 | + +## Model Latency Comparison + +| Model | Avg (s) | Min (s) | Max (s) | P50 (s) | P95 (s) | +|-------|---------|---------|---------|---------|---------| +| sonnet-4.5 | 40.5 | 40.5 | 40.5 | 40.5 | 40.5 | + +## Performance by Tag + +Success rate by test category and model: + +| Tag | sonnet-4.5 | Warnings | +|-----|-------|----------| +| [easy](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522easy%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520easy%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | +| [kubernetes](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522kubernetes%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520kubernetes%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | +| [one-test](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522one-test%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520one-test%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | +| [regression](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522regression%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520regression%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | +| **Overall** | 🟢 100% (1/1) | | + +## Raw Results + +Status of all evaluations across models. Color coding: + +- 🟢 Passing 100% (stable) +- 🟡 Passing 1-99% +- 🔴 Passing 0% (failing) +- 🔧 Mock data failure (missing or invalid test data) +- ⚠️ Setup failure (environment/infrastructure issue) +- ⏱️ Timeout or rate limit error +- ⏭️ Test skipped (e.g., known issue or precondition not met) + +| Eval ID | [sonnet-4.5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522sonnet-4.5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520sonnet-4.5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +|---------|-------| +| [**09_crashpod**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +| **SUMMARY** | 🟢 100% (1/1) | + +## Detailed Raw Results + +| Eval ID | sonnet-4.5 | +|---------|-------| +| [09_crashpod](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 40.5s / 💰 $0.18 | + +--- +*Results are automatically generated and updated weekly. View full traces and detailed analysis in [Braintrust experiment: local-benchmark-20260104-174426](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426).* diff --git a/docs/development/evaluations/history/fast/.gitkeep b/docs/development/evaluations/history/fast/.gitkeep new file mode 100644 index 0000000000..e69de29bb2 diff --git a/docs/development/evaluations/history/.nav.yml b/docs/development/evaluations/history/fast/.nav.yml similarity index 100% rename from docs/development/evaluations/history/.nav.yml rename to docs/development/evaluations/history/fast/.nav.yml diff --git a/docs/development/evaluations/history/fast/index.md b/docs/development/evaluations/history/fast/index.md new file mode 100644 index 0000000000..2d54f2f083 --- /dev/null +++ b/docs/development/evaluations/history/fast/index.md @@ -0,0 +1,12 @@ +# Fast Benchmark Results + +!!! info "Fast Benchmark" + **Markers**: `regression or benchmark`
+ **Schedule**: Weekly (Sunday 2 AM UTC)
+ **Purpose**: Quick regression tests to catch breaking changes + +Fast benchmarks run a subset of critical tests designed to complete quickly while catching regressions. These are the primary automated benchmarks that run weekly. + +## Recent Results + +Browse the results below to track fast benchmark performance over time. diff --git a/docs/development/evaluations/history/fast/results_20260104_174301.md b/docs/development/evaluations/history/fast/results_20260104_174301.md new file mode 100644 index 0000000000..db38c0879c --- /dev/null +++ b/docs/development/evaluations/history/fast/results_20260104_174301.md @@ -0,0 +1,121 @@ +# January 04, 2026 + +**Generated**: 2026-01-04 17:43 UTC +**Total Duration**: 43m 0s +**Iterations**: 5 +**Judge (classifier) model**: gpt-4.1 + +## About this Benchmark + +**Fast Benchmark**: Quick regression tests using markers `regression or benchmark` - designed to run frequently and catch regressions. + +HolmesGPT is continuously evaluated against real-world Kubernetes and cloud troubleshooting scenarios. + +If you find scenarios that HolmesGPT does not perform well on, please consider adding them as evals to the benchmark. + +## Model Accuracy Comparison + +| Model | Pass | Fail | Skip/Error | Total | Success Rate | +|-------|------|------|------------|-------|--------------| +| deepseek-3.1 | 49 | 21 | 0 | 70 | 🟡 70% (49/70) | +| gpt-5 | 37 | 33 | 0 | 70 | 🟡 53% (37/70) | +| gpt-5.1 | 36 | 34 | 0 | 70 | 🟡 51% (36/70) | +| haiku-4.5 | 44 | 26 | 0 | 70 | 🟡 63% (44/70) | +| sonnet-4.5 | 59 | 11 | 0 | 70 | 🟡 84% (59/70) | + +## Model Cost Comparison + +| Model | Tests | Avg Cost | Min Cost | Max Cost | Total Cost | +|-------|-------|----------|----------|----------|------------| +| gpt-5 | 70 | $0.06 | $0.00 | $0.28 | $4.38 | +| gpt-5.1 | 65 | $0.11 | $0.00 | $0.39 | $7.46 | +| haiku-4.5 | 70 | $0.04 | $0.00 | $0.11 | $2.84 | +| sonnet-4.5 | 70 | $0.17 | $0.01 | $0.34 | $11.66 | + +## Model Latency Comparison + +| Model | Avg (s) | Min (s) | Max (s) | P50 (s) | P95 (s) | +|-------|---------|---------|---------|---------|---------| +| deepseek-3.1 | 87.8 | 6.6 | 167.8 | 93.2 | 143.1 | +| gpt-5 | 33.1 | 3.7 | 639.3 | 25.7 | 46.6 | +| gpt-5.1 | 89.1 | 6.5 | 338.4 | 74.4 | 202.3 | +| haiku-4.5 | 27.5 | 3.1 | 59.8 | 30.4 | 42.2 | +| sonnet-4.5 | 43.6 | 7.6 | 94.5 | 46.6 | 66.2 | + +## Performance by Tag + +Success rate by test category and model: + +| Tag | deepseek-3.1 | gpt-5 | gpt-5.1 | haiku-4.5 | sonnet-4.5 | Warnings | +|-----|-------|-------|-------|-------|-------|----------| +| [benchmark](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522benchmark%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520benchmark%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟡 44% (11/25) | 🟡 48% (12/25) | 🟡 36% (9/25) | 🟡 28% (7/25) | 🟡 56% (14/25) | | +| [context_window](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522context_window%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520context_window%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟡 90% (9/10) | 🟡 90% (9/10) | 🟡 50% (5/10) | 🟡 20% (2/10) | 🟡 90% (9/10) | | +| [counting](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522counting%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520counting%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (5/5) | 🟢 100% (5/5) | 🟢 100% (5/5) | 🟢 100% (5/5) | 🟢 100% (5/5) | | +| [datetime](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522datetime%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520datetime%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟡 87% (13/15) | 🟡 87% (13/15) | 🟡 67% (10/15) | 🟡 47% (7/15) | 🟡 93% (14/15) | | +| [easy](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522easy%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520easy%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟡 89% (31/35) | 🟡 63% (22/35) | 🟡 66% (23/35) | 🟡 86% (30/35) | 🟢 100% (35/35) | | +| [grafana-dashboard](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522grafana-dashboard%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520grafana-dashboard%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🔴 0% (0/5) | 🔴 0% (0/5) | 🔴 0% (0/5) | 🔴 0% (0/5) | 🔴 0% (0/5) | | +| [hard](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522hard%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520hard%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🔴 0% (0/5) | 🔴 0% (0/5) | 🔴 0% (0/5) | 🔴 0% (0/5) | 🔴 0% (0/5) | | +| [kubernetes](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522kubernetes%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520kubernetes%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟡 71% (25/35) | 🟡 43% (15/35) | 🟡 46% (16/35) | 🟡 63% (22/35) | 🟡 86% (30/35) | | +| [logs](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522logs%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520logs%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟡 60% (12/20) | 🟡 70% (14/20) | 🟡 50% (10/20) | 🟡 35% (7/20) | 🟡 70% (14/20) | | +| [loki](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522loki%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520loki%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟡 60% (3/5) | 🟢 100% (5/5) | 🟢 100% (5/5) | 🟡 80% (4/5) | 🟢 100% (5/5) | | +| [medium](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522medium%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520medium%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟡 60% (18/30) | 🟡 50% (15/30) | 🟡 43% (13/30) | 🟡 47% (14/30) | 🟡 80% (24/30) | | +| [metrics](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522metrics%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520metrics%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🔴 0% (0/5) | 🔴 0% (0/5) | 🔴 0% (0/5) | 🔴 0% (0/5) | 🔴 0% (0/5) | | +| [network](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522network%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520network%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (5/5) | 🟢 100% (5/5) | 🟡 80% (4/5) | 🟢 100% (5/5) | 🟢 100% (5/5) | | +| [one-test](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522one-test%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520one-test%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (5/5) | 🟡 40% (2/5) | 🟡 60% (3/5) | 🟢 100% (5/5) | 🟢 100% (5/5) | | +| [port-forward](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522port-forward%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520port-forward%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟡 30% (3/10) | 🟡 50% (5/10) | 🟡 50% (5/10) | 🟡 40% (4/10) | 🟡 50% (5/10) | | +| [question-answer](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522question-answer%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520question-answer%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🔴 0% (0/5) | 🔴 0% (0/5) | 🔴 0% (0/5) | 🔴 0% (0/5) | 🔴 0% (0/5) | | +| [regression](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522regression%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520regression%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟡 84% (38/45) | 🟡 56% (25/45) | 🟡 60% (27/45) | 🟡 82% (37/45) | 🟢 100% (45/45) | | +| [runbooks](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522runbooks%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520runbooks%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟡 40% (4/10) | 🟡 60% (6/10) | 🟡 60% (6/10) | 🟡 70% (7/10) | 🟢 100% (10/10) | | +| **Overall** | 🟡 70% (49/70) | 🟡 53% (37/70) | 🟡 51% (36/70) | 🟡 63% (44/70) | 🟡 84% (59/70) | | + +## Raw Results + +Status of all evaluations across models. Color coding: + +- 🟢 Passing 100% (stable) +- 🟡 Passing 1-99% +- 🔴 Passing 0% (failing) +- 🔧 Mock data failure (missing or invalid test data) +- ⚠️ Setup failure (environment/infrastructure issue) +- ⏱️ Timeout or rate limit error +- ⏭️ Test skipped (e.g., known issue or precondition not met) + +| Eval ID | [deepseek-3.1](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522deepseek-3.1%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520deepseek-3.1%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [gpt-5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522gpt-5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520gpt-5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [gpt-5.1](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522gpt-5.1%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520gpt-5.1%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [haiku-4.5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522haiku-4.5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520haiku-4.5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [sonnet-4.5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522sonnet-4.5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520sonnet-4.5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +|---------|-------|-------|-------|-------|-------| +| [**09_crashpod**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +| [**101_loki_historical_logs_pod_deleted**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/101_loki_historical_logs_pod_deleted/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%2522101_loki_historical_logs_pod_deleted%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%2520101_loki_historical_logs_pod_deleted%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522101_loki_historical_logs_pod_deleted%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520101_loki_historical_logs_pod_deleted%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522101_loki_historical_logs_pod_deleted%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520101_loki_historical_logs_pod_deleted%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522101_loki_historical_logs_pod_deleted%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520101_loki_historical_logs_pod_deleted%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522101_loki_historical_logs_pod_deleted%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520101_loki_historical_logs_pod_deleted%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522101_loki_historical_logs_pod_deleted%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520101_loki_historical_logs_pod_deleted%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +| [**108_logs_nearby_lines**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/108_logs_nearby_lines/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%2522108_logs_nearby_lines%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%2520108_logs_nearby_lines%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522108_logs_nearby_lines%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520108_logs_nearby_lines%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522108_logs_nearby_lines%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520108_logs_nearby_lines%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522108_logs_nearby_lines%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520108_logs_nearby_lines%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522108_logs_nearby_lines%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520108_logs_nearby_lines%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522108_logs_nearby_lines%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520108_logs_nearby_lines%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +| [**111_pod_names_contain_service**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/111_pod_names_contain_service/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%2522111_pod_names_contain_service%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%2520111_pod_names_contain_service%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522111_pod_names_contain_service%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520111_pod_names_contain_service%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522111_pod_names_contain_service%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520111_pod_names_contain_service%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522111_pod_names_contain_service%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520111_pod_names_contain_service%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522111_pod_names_contain_service%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520111_pod_names_contain_service%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522111_pod_names_contain_service%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520111_pod_names_contain_service%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +| [**12_job_crashing**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/12_job_crashing/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252212_job_crashing%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252012_job_crashing%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252212_job_crashing%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252012_job_crashing%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252212_job_crashing%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252012_job_crashing%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252212_job_crashing%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252012_job_crashing%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252212_job_crashing%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252012_job_crashing%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252212_job_crashing%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252012_job_crashing%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +| [**162_get_runbooks**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/162_get_runbooks/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%2522162_get_runbooks%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%2520162_get_runbooks%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522162_get_runbooks%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520162_get_runbooks%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522162_get_runbooks%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520162_get_runbooks%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522162_get_runbooks%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520162_get_runbooks%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522162_get_runbooks%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520162_get_runbooks%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522162_get_runbooks%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520162_get_runbooks%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +| [**176_network_policy_blocking_traffic_no_runbooks**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/176_network_policy_blocking_traffic_no_runbooks/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%2522176_network_policy_blocking_traffic_no_runbooks%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%2520176_network_policy_blocking_traffic_no_runbooks%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522176_network_policy_blocking_traffic_no_runbooks%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520176_network_policy_blocking_traffic_no_runbooks%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522176_network_policy_blocking_traffic_no_runbooks%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520176_network_policy_blocking_traffic_no_runbooks%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522176_network_policy_blocking_traffic_no_runbooks%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520176_network_policy_blocking_traffic_no_runbooks%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522176_network_policy_blocking_traffic_no_runbooks%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520176_network_policy_blocking_traffic_no_runbooks%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522176_network_policy_blocking_traffic_no_runbooks%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520176_network_policy_blocking_traffic_no_runbooks%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +| [**179_grafana_big_dashboard_query**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/179_grafana_big_dashboard_query/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%2522179_grafana_big_dashboard_query%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%2520179_grafana_big_dashboard_query%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522179_grafana_big_dashboard_query%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520179_grafana_big_dashboard_query%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522179_grafana_big_dashboard_query%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520179_grafana_big_dashboard_query%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522179_grafana_big_dashboard_query%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520179_grafana_big_dashboard_query%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522179_grafana_big_dashboard_query%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520179_grafana_big_dashboard_query%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522179_grafana_big_dashboard_query%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520179_grafana_big_dashboard_query%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +| [**24_misconfigured_pvc**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/24_misconfigured_pvc/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252224_misconfigured_pvc%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252024_misconfigured_pvc%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252224_misconfigured_pvc%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252024_misconfigured_pvc%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252224_misconfigured_pvc%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252024_misconfigured_pvc%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252224_misconfigured_pvc%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252024_misconfigured_pvc%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252224_misconfigured_pvc%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252024_misconfigured_pvc%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252224_misconfigured_pvc%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252024_misconfigured_pvc%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +| [**43_current_datetime_from_prompt**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/43_current_datetime_from_prompt/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252243_current_datetime_from_prompt%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252043_current_datetime_from_prompt%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252243_current_datetime_from_prompt%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252043_current_datetime_from_prompt%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252243_current_datetime_from_prompt%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252043_current_datetime_from_prompt%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252243_current_datetime_from_prompt%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252043_current_datetime_from_prompt%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252243_current_datetime_from_prompt%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252043_current_datetime_from_prompt%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252243_current_datetime_from_prompt%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252043_current_datetime_from_prompt%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +| [**61_exact_match_counting**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/61_exact_match_counting/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252261_exact_match_counting%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252061_exact_match_counting%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252261_exact_match_counting%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252061_exact_match_counting%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252261_exact_match_counting%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252061_exact_match_counting%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252261_exact_match_counting%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252061_exact_match_counting%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252261_exact_match_counting%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252061_exact_match_counting%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252261_exact_match_counting%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252061_exact_match_counting%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +| [**73a_time_window_anomaly**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/73a_time_window_anomaly/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252273a_time_window_anomaly%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252073a_time_window_anomaly%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273a_time_window_anomaly%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073a_time_window_anomaly%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273a_time_window_anomaly%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073a_time_window_anomaly%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273a_time_window_anomaly%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073a_time_window_anomaly%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273a_time_window_anomaly%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073a_time_window_anomaly%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273a_time_window_anomaly%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073a_time_window_anomaly%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +| [**73b_time_window_anomaly**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/73b_time_window_anomaly/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252273b_time_window_anomaly%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252073b_time_window_anomaly%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273b_time_window_anomaly%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073b_time_window_anomaly%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273b_time_window_anomaly%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073b_time_window_anomaly%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273b_time_window_anomaly%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073b_time_window_anomaly%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273b_time_window_anomaly%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073b_time_window_anomaly%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273b_time_window_anomaly%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073b_time_window_anomaly%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +| [**96_no_matching_runbook**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/96_no_matching_runbook/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252296_no_matching_runbook%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252096_no_matching_runbook%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252296_no_matching_runbook%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252096_no_matching_runbook%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252296_no_matching_runbook%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252096_no_matching_runbook%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252296_no_matching_runbook%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252096_no_matching_runbook%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252296_no_matching_runbook%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252096_no_matching_runbook%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252296_no_matching_runbook%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252096_no_matching_runbook%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +| **SUMMARY** | 🟡 70% (49/70) | 🟡 53% (37/70) | 🟡 51% (36/70) | 🟡 63% (44/70) | 🟡 84% (59/70) | + +## Detailed Raw Results + +| Eval ID | deepseek-3.1 | gpt-5 | gpt-5.1 | haiku-4.5 | sonnet-4.5 | +|---------|-------|-------|-------|-------|-------| +| [09_crashpod](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 83.5s | [🟡 40% (2/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 15.1s / 💰 $0.04 | [🟡 60% (3/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 49.8s / 💰 $0.06 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 26.4s / 💰 $0.03 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 38.1s / 💰 $0.13 | +| [101_loki_historical_logs_pod_deleted](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/101_loki_historical_logs_pod_deleted/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%2522101_loki_historical_logs_pod_deleted%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%2520101_loki_historical_logs_pod_deleted%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡 60% (3/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522101_loki_historical_logs_pod_deleted%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520101_loki_historical_logs_pod_deleted%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 110.9s | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522101_loki_historical_logs_pod_deleted%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520101_loki_historical_logs_pod_deleted%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 29.9s / 💰 $0.08 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522101_loki_historical_logs_pod_deleted%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520101_loki_historical_logs_pod_deleted%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 196.8s / 💰 $0.24 | [🟡 80% (4/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522101_loki_historical_logs_pod_deleted%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520101_loki_historical_logs_pod_deleted%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 36.0s / 💰 $0.05 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522101_loki_historical_logs_pod_deleted%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520101_loki_historical_logs_pod_deleted%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 45.9s / 💰 $0.17 | +| [108_logs_nearby_lines](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/108_logs_nearby_lines/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%2522108_logs_nearby_lines%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%2520108_logs_nearby_lines%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴 0% (0/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522108_logs_nearby_lines%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520108_logs_nearby_lines%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 134.4s | [🔴 0% (0/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522108_logs_nearby_lines%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520108_logs_nearby_lines%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 33.2s / 💰 $0.09 | [🔴 0% (0/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522108_logs_nearby_lines%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520108_logs_nearby_lines%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 131.9s / 💰 $0.18 | [🟡 20% (1/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522108_logs_nearby_lines%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520108_logs_nearby_lines%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 36.5s / 💰 $0.06 | [🔴 0% (0/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522108_logs_nearby_lines%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520108_logs_nearby_lines%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 63.4s / 💰 $0.21 | +| [111_pod_names_contain_service](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/111_pod_names_contain_service/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%2522111_pod_names_contain_service%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%2520111_pod_names_contain_service%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522111_pod_names_contain_service%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520111_pod_names_contain_service%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 95.7s | [🔴 0% (0/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522111_pod_names_contain_service%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520111_pod_names_contain_service%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 12.4s / 💰 $0.02 | [🟡 40% (2/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522111_pod_names_contain_service%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520111_pod_names_contain_service%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 67.5s / 💰 $0.07 | [🟡 80% (4/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522111_pod_names_contain_service%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520111_pod_names_contain_service%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 24.0s / 💰 $0.03 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522111_pod_names_contain_service%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520111_pod_names_contain_service%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 53.5s / 💰 $0.23 | +| [12_job_crashing](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/12_job_crashing/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252212_job_crashing%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252012_job_crashing%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡 80% (4/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252212_job_crashing%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252012_job_crashing%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 95.0s | [🟡 20% (1/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252212_job_crashing%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252012_job_crashing%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 31.4s / 💰 $0.07 | [🟡 20% (1/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252212_job_crashing%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252012_job_crashing%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 85.4s / 💰 $0.11 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252212_job_crashing%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252012_job_crashing%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 31.0s / 💰 $0.04 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252212_job_crashing%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252012_job_crashing%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 48.8s / 💰 $0.16 | +| [162_get_runbooks](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/162_get_runbooks/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%2522162_get_runbooks%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%2520162_get_runbooks%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡 40% (2/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522162_get_runbooks%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520162_get_runbooks%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 82.2s | [🟡 60% (3/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522162_get_runbooks%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520162_get_runbooks%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 159.8s / 💰 $0.13 | [🟡 40% (2/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522162_get_runbooks%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520162_get_runbooks%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 96.1s / 💰 $0.12 | [🟡 60% (3/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522162_get_runbooks%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520162_get_runbooks%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 36.0s / 💰 $0.07 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522162_get_runbooks%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520162_get_runbooks%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 50.5s / 💰 $0.21 | +| [176_network_policy_blocking_traffic_no_runbooks](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/176_network_policy_blocking_traffic_no_runbooks/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%2522176_network_policy_blocking_traffic_no_runbooks%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%2520176_network_policy_blocking_traffic_no_runbooks%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522176_network_policy_blocking_traffic_no_runbooks%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520176_network_policy_blocking_traffic_no_runbooks%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 93.0s | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522176_network_policy_blocking_traffic_no_runbooks%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520176_network_policy_blocking_traffic_no_runbooks%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 35.3s / 💰 $0.09 | [🟡 80% (4/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522176_network_policy_blocking_traffic_no_runbooks%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520176_network_policy_blocking_traffic_no_runbooks%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 91.9s / 💰 $0.11 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522176_network_policy_blocking_traffic_no_runbooks%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520176_network_policy_blocking_traffic_no_runbooks%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 37.8s / 💰 $0.06 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522176_network_policy_blocking_traffic_no_runbooks%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520176_network_policy_blocking_traffic_no_runbooks%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 49.4s / 💰 $0.21 | +| [179_grafana_big_dashboard_query](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/179_grafana_big_dashboard_query/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%2522179_grafana_big_dashboard_query%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%2520179_grafana_big_dashboard_query%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴 0% (0/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522179_grafana_big_dashboard_query%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520179_grafana_big_dashboard_query%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 94.4s | [🔴 0% (0/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522179_grafana_big_dashboard_query%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520179_grafana_big_dashboard_query%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 21.2s / 💰 $0.04 | [🔴 0% (0/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522179_grafana_big_dashboard_query%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520179_grafana_big_dashboard_query%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 175.0s / 💰 $0.20 | [🔴 0% (0/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522179_grafana_big_dashboard_query%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520179_grafana_big_dashboard_query%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 29.5s / 💰 $0.04 | [🔴 0% (0/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522179_grafana_big_dashboard_query%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520179_grafana_big_dashboard_query%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 34.4s / 💰 $0.13 | +| [24_misconfigured_pvc](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/24_misconfigured_pvc/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252224_misconfigured_pvc%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252024_misconfigured_pvc%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252224_misconfigured_pvc%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252024_misconfigured_pvc%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 103.6s | [🔴 0% (0/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252224_misconfigured_pvc%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252024_misconfigured_pvc%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 4.3s / 💰 $0.00 | [🔴 0% (0/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252224_misconfigured_pvc%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252024_misconfigured_pvc%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 16.9s / 💰 $0.01 | [🟡 20% (1/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252224_misconfigured_pvc%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252024_misconfigured_pvc%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 9.9s / 💰 $0.01 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252224_misconfigured_pvc%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252024_misconfigured_pvc%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 42.5s / 💰 $0.14 | +| [43_current_datetime_from_prompt](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/43_current_datetime_from_prompt/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252243_current_datetime_from_prompt%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252043_current_datetime_from_prompt%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡 80% (4/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252243_current_datetime_from_prompt%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252043_current_datetime_from_prompt%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 12.9s | [🟡 80% (4/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252243_current_datetime_from_prompt%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252043_current_datetime_from_prompt%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 7.4s / 💰 $0.01 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252243_current_datetime_from_prompt%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252043_current_datetime_from_prompt%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 24.0s / 💰 $0.02 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252243_current_datetime_from_prompt%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252043_current_datetime_from_prompt%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 6.3s / 💰 $0.01 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252243_current_datetime_from_prompt%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252043_current_datetime_from_prompt%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 9.8s / 💰 $0.01 | +| [61_exact_match_counting](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/61_exact_match_counting/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252261_exact_match_counting%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252061_exact_match_counting%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252261_exact_match_counting%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252061_exact_match_counting%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 35.6s | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252261_exact_match_counting%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252061_exact_match_counting%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 16.4s / 💰 $0.04 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252261_exact_match_counting%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252061_exact_match_counting%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 49.8s / 💰 $0.06 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252261_exact_match_counting%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252061_exact_match_counting%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 14.0s / 💰 $0.01 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252261_exact_match_counting%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252061_exact_match_counting%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 19.1s / 💰 $0.06 | +| [73a_time_window_anomaly](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/73a_time_window_anomaly/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252273a_time_window_anomaly%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252073a_time_window_anomaly%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡 80% (4/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273a_time_window_anomaly%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073a_time_window_anomaly%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 95.6s | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273a_time_window_anomaly%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073a_time_window_anomaly%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 27.5s / 💰 $0.07 | [🟡 80% (4/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273a_time_window_anomaly%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073a_time_window_anomaly%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 93.0s / 💰 $0.13 | [🟡 20% (1/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273a_time_window_anomaly%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073a_time_window_anomaly%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 31.8s / 💰 $0.05 | [🟡 80% (4/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273a_time_window_anomaly%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073a_time_window_anomaly%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 48.6s / 💰 $0.20 | +| [73b_time_window_anomaly](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/73b_time_window_anomaly/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252273b_time_window_anomaly%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252073b_time_window_anomaly%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273b_time_window_anomaly%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073b_time_window_anomaly%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 93.2s | [🟡 80% (4/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273b_time_window_anomaly%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073b_time_window_anomaly%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 28.2s / 💰 $0.07 | [🟡 20% (1/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273b_time_window_anomaly%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073b_time_window_anomaly%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 37.3s / 💰 $0.03 | [🟡 20% (1/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273b_time_window_anomaly%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073b_time_window_anomaly%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 32.9s / 💰 $0.05 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273b_time_window_anomaly%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073b_time_window_anomaly%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 46.3s / 💰 $0.18 | +| [96_no_matching_runbook](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/96_no_matching_runbook/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252296_no_matching_runbook%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252096_no_matching_runbook%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡 40% (2/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252296_no_matching_runbook%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252096_no_matching_runbook%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 99.9s | [🟡 60% (3/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252296_no_matching_runbook%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252096_no_matching_runbook%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 41.0s / 💰 $0.13 | [🟡 80% (4/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252296_no_matching_runbook%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252096_no_matching_runbook%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 132.4s / 💰 $0.17 | [🟡 80% (4/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252296_no_matching_runbook%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252096_no_matching_runbook%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 33.5s / 💰 $0.06 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252296_no_matching_runbook%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252096_no_matching_runbook%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 60.6s / 💰 $0.30 | + +--- +*Results are automatically generated and updated weekly. View full traces and detailed analysis in [Braintrust experiment: local-benchmark-20260101-140005](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005).* diff --git a/docs/development/evaluations/history/fast/results_20260104_174350.md b/docs/development/evaluations/history/fast/results_20260104_174350.md new file mode 100644 index 0000000000..6237ae47eb --- /dev/null +++ b/docs/development/evaluations/history/fast/results_20260104_174350.md @@ -0,0 +1,22 @@ +# January 04, 2026 + +**Generated**: 2026-01-04 17:43 UTC +**Total Duration**: 10s +**Iterations**: 1 +**Judge (classifier) model**: gpt-4.1 + +## About this Benchmark + +**Fast Benchmark**: Quick regression tests using markers `regression or benchmark` - designed to run frequently and catch regressions. + +HolmesGPT is continuously evaluated against real-world Kubernetes and cloud troubleshooting scenarios. + +If you find scenarios that HolmesGPT does not perform well on, please consider adding them as evals to the benchmark. + +## Model Accuracy Comparison + +| Model | Pass | Fail | Skip/Error | Total | Success Rate | +|-------|------|------|------------|-------|--------------| + +--- +*Results are automatically generated and updated weekly. For detailed traces and analysis, see our [Braintrust dashboard](https://braintrust.dev).* diff --git a/docs/development/evaluations/history/fast/results_20260104_174620.md b/docs/development/evaluations/history/fast/results_20260104_174620.md new file mode 100644 index 0000000000..6285c1a4f0 --- /dev/null +++ b/docs/development/evaluations/history/fast/results_20260104_174620.md @@ -0,0 +1,70 @@ +# January 04, 2026 + +**Generated**: 2026-01-04 17:46 UTC +**Total Duration**: 1m 18s +**Iterations**: 1 +**Judge (classifier) model**: gpt-4.1 + +## About this Benchmark + +**Fast Benchmark**: Quick regression tests using markers `regression or benchmark` - designed to run frequently and catch regressions. + +HolmesGPT is continuously evaluated against real-world Kubernetes and cloud troubleshooting scenarios. + +If you find scenarios that HolmesGPT does not perform well on, please consider adding them as evals to the benchmark. + +## Model Accuracy Comparison + +| Model | Pass | Fail | Skip/Error | Total | Success Rate | +|-------|------|------|------------|-------|--------------| +| sonnet-4.5 | 1 | 0 | 0 | 1 | 🟢 100% (1/1) | + +## Model Cost Comparison + +| Model | Tests | Avg Cost | Min Cost | Max Cost | Total Cost | +|-------|-------|----------|----------|----------|------------| +| sonnet-4.5 | 1 | $0.18 | $0.18 | $0.18 | $0.18 | + +## Model Latency Comparison + +| Model | Avg (s) | Min (s) | Max (s) | P50 (s) | P95 (s) | +|-------|---------|---------|---------|---------|---------| +| sonnet-4.5 | 40.5 | 40.5 | 40.5 | 40.5 | 40.5 | + +## Performance by Tag + +Success rate by test category and model: + +| Tag | sonnet-4.5 | Warnings | +|-----|-------|----------| +| [easy](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522easy%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520easy%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | +| [kubernetes](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522kubernetes%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520kubernetes%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | +| [one-test](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522one-test%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520one-test%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | +| [regression](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522regression%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520regression%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | +| **Overall** | 🟢 100% (1/1) | | + +## Raw Results + +Status of all evaluations across models. Color coding: + +- 🟢 Passing 100% (stable) +- 🟡 Passing 1-99% +- 🔴 Passing 0% (failing) +- 🔧 Mock data failure (missing or invalid test data) +- ⚠️ Setup failure (environment/infrastructure issue) +- ⏱️ Timeout or rate limit error +- ⏭️ Test skipped (e.g., known issue or precondition not met) + +| Eval ID | [sonnet-4.5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522sonnet-4.5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520sonnet-4.5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +|---------|-------| +| [**09_crashpod**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +| **SUMMARY** | 🟢 100% (1/1) | + +## Detailed Raw Results + +| Eval ID | sonnet-4.5 | +|---------|-------| +| [09_crashpod](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 40.5s / 💰 $0.18 | + +--- +*Results are automatically generated and updated weekly. View full traces and detailed analysis in [Braintrust experiment: local-benchmark-20260104-174426](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426).* diff --git a/docs/development/evaluations/history/full/.gitkeep b/docs/development/evaluations/history/full/.gitkeep new file mode 100644 index 0000000000..e69de29bb2 diff --git a/docs/development/evaluations/history/full/.nav.yml b/docs/development/evaluations/history/full/.nav.yml new file mode 100644 index 0000000000..d71105aebd --- /dev/null +++ b/docs/development/evaluations/history/full/.nav.yml @@ -0,0 +1,5 @@ +sort: + direction: desc +nav: + - index.md + - "*" diff --git a/docs/development/evaluations/history/custom_claude_results_20250930_153753.md b/docs/development/evaluations/history/full/custom_claude_results_20250930_153753.md similarity index 100% rename from docs/development/evaluations/history/custom_claude_results_20250930_153753.md rename to docs/development/evaluations/history/full/custom_claude_results_20250930_153753.md diff --git a/docs/development/evaluations/history/custom_self_hosted_results_20251008_053744.md b/docs/development/evaluations/history/full/custom_self_hosted_results_20251008_053744.md similarity index 100% rename from docs/development/evaluations/history/custom_self_hosted_results_20251008_053744.md rename to docs/development/evaluations/history/full/custom_self_hosted_results_20251008_053744.md diff --git a/docs/development/evaluations/history/full/index.md b/docs/development/evaluations/history/full/index.md new file mode 100644 index 0000000000..4d284a7722 --- /dev/null +++ b/docs/development/evaluations/history/full/index.md @@ -0,0 +1,16 @@ +# Full Benchmark Results + +!!! info "Full Benchmark" + **Markers**: `easy or medium or hard or regression or benchmark`
+ **Schedule**: Manual / On-demand
+ **Purpose**: Comprehensive testing across all difficulty levels + +Full benchmarks run the complete test suite including easy, medium, and hard difficulty tests. These provide comprehensive coverage but take longer to complete. + +## Historical Results + +This section includes: + +- **Full benchmark runs** - Comprehensive test suite executions +- **Extended comparisons** - Special runs comparing multiple models +- **Legacy results** - Historical benchmark data from before the fast/full separation diff --git a/docs/development/evaluations/history/results_20250928_001434.md b/docs/development/evaluations/history/full/results_20250928_001434.md similarity index 100% rename from docs/development/evaluations/history/results_20250928_001434.md rename to docs/development/evaluations/history/full/results_20250928_001434.md diff --git a/docs/development/evaluations/history/results_20250930_085923.md b/docs/development/evaluations/history/full/results_20250930_085923.md similarity index 100% rename from docs/development/evaluations/history/results_20250930_085923.md rename to docs/development/evaluations/history/full/results_20250930_085923.md diff --git a/docs/development/evaluations/history/results_20251012_170303.md b/docs/development/evaluations/history/full/results_20251012_170303.md similarity index 100% rename from docs/development/evaluations/history/results_20251012_170303.md rename to docs/development/evaluations/history/full/results_20251012_170303.md diff --git a/docs/development/evaluations/history/results_20251127_042958.md b/docs/development/evaluations/history/full/results_20251127_042958.md similarity index 100% rename from docs/development/evaluations/history/results_20251127_042958.md rename to docs/development/evaluations/history/full/results_20251127_042958.md diff --git a/docs/development/evaluations/history/index.md b/docs/development/evaluations/history/index.md deleted file mode 100644 index 52ad48e0fc..0000000000 --- a/docs/development/evaluations/history/index.md +++ /dev/null @@ -1,9 +0,0 @@ -# Historical Evaluation Results - -Browse through our past benchmark runs to track performance trends over time. - -## Weekly Results -Regular weekly benchmark runs that track model performance over time. - -## Extended Comparisons -Special benchmark runs comparing multiple models and configurations. diff --git a/docs/development/evaluations/index.md b/docs/development/evaluations/index.md index f24b4e406f..9d1c8ac1e6 100644 --- a/docs/development/evaluations/index.md +++ b/docs/development/evaluations/index.md @@ -6,10 +6,21 @@ We also use the evals as regression tests on every commit. **[View latest evaluation results →](./latest-results.md)** +## Benchmark Types + +We run two types of benchmarks to balance speed and coverage: + +| Benchmark | Markers | Purpose | Schedule | +|-----------|---------|---------|----------| +| **[Fast Benchmark](./history/fast/index.md)** | `regression or benchmark` | Quick regression tests to catch breaking changes | Weekly (Sunday 2 AM UTC) | +| **[Full Benchmark](./history/full/index.md)** | `easy or medium or hard or regression or benchmark` | Comprehensive testing across all difficulty levels | Manual / On-demand | + ## Test Categories -- **Regression tests (`easy`)**: Scenarios that must always pass -- **Advanced tests (`medium` and `hard`)**: More challenging scenarios +- **Regression tests (`regression`, `benchmark`)**: Critical scenarios that must always pass +- **Easy tests (`easy`)**: Straightforward scenarios for baseline validation +- **Medium tests (`medium`)**: Moderately complex troubleshooting scenarios +- **Hard tests (`hard`)**: Challenging multi-step investigations - **Specialized tests**: Focused on specific capabilities (logs, kubernetes, prometheus, etc.) ## Quick Start diff --git a/docs/development/evaluations/latest-results.md b/docs/development/evaluations/latest-results.md index 4be50c00b0..39f863ff4a 100644 --- a/docs/development/evaluations/latest-results.md +++ b/docs/development/evaluations/latest-results.md @@ -1,11 +1,14 @@ -# HolmesGPT LLM Evaluation Benchmark Results +# HolmesGPT LLM Evaluation Fast Benchmark Results -**Generated**: 2026-01-01 14:44 UTC -**Total Duration**: 43m 0s -**Iterations**: 5 +**Generated**: 2026-01-04 18:10 UTC +**Total Duration**: 1m 18s +**Iterations**: 1 **Judge (classifier) model**: gpt-4.1 -## About this Benchmark +!!! info "Fast Benchmark" + **Markers**: `regression or benchmark`
+ **Schedule**: Weekly (Sunday 2 AM UTC)
+ **Purpose**: Quick regression tests to catch breaking changes HolmesGPT is continuously evaluated against real-world Kubernetes and cloud troubleshooting scenarios. @@ -15,56 +18,31 @@ If you find scenarios that HolmesGPT does not perform well on, please consider a | Model | Pass | Fail | Skip/Error | Total | Success Rate | |-------|------|------|------------|-------|--------------| -| deepseek-3.1 | 49 | 21 | 0 | 70 | 🟡 70% (49/70) | -| gpt-5 | 37 | 33 | 0 | 70 | 🟡 53% (37/70) | -| gpt-5.1 | 36 | 34 | 0 | 70 | 🟡 51% (36/70) | -| haiku-4.5 | 44 | 26 | 0 | 70 | 🟡 63% (44/70) | -| sonnet-4.5 | 59 | 11 | 0 | 70 | 🟡 84% (59/70) | +| sonnet-4.5 | 1 | 0 | 0 | 1 | 🟢 100% (1/1) | ## Model Cost Comparison | Model | Tests | Avg Cost | Min Cost | Max Cost | Total Cost | |-------|-------|----------|----------|----------|------------| -| gpt-5 | 70 | $0.06 | $0.00 | $0.28 | $4.38 | -| gpt-5.1 | 65 | $0.11 | $0.00 | $0.39 | $7.46 | -| haiku-4.5 | 70 | $0.04 | $0.00 | $0.11 | $2.84 | -| sonnet-4.5 | 70 | $0.17 | $0.01 | $0.34 | $11.66 | +| sonnet-4.5 | 1 | $0.18 | $0.18 | $0.18 | $0.18 | ## Model Latency Comparison | Model | Avg (s) | Min (s) | Max (s) | P50 (s) | P95 (s) | |-------|---------|---------|---------|---------|---------| -| deepseek-3.1 | 87.8 | 6.6 | 167.8 | 93.2 | 143.1 | -| gpt-5 | 33.1 | 3.7 | 639.3 | 25.7 | 46.6 | -| gpt-5.1 | 89.1 | 6.5 | 338.4 | 74.4 | 202.3 | -| haiku-4.5 | 27.5 | 3.1 | 59.8 | 30.4 | 42.2 | -| sonnet-4.5 | 43.6 | 7.6 | 94.5 | 46.6 | 66.2 | +| sonnet-4.5 | 40.5 | 40.5 | 40.5 | 40.5 | 40.5 | ## Performance by Tag Success rate by test category and model: -| Tag | deepseek-3.1 | gpt-5 | gpt-5.1 | haiku-4.5 | sonnet-4.5 | Warnings | -|-----|-------|-------|-------|-------|-------|----------| -| [benchmark](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522benchmark%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520benchmark%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟡 44% (11/25) | 🟡 48% (12/25) | 🟡 36% (9/25) | 🟡 28% (7/25) | 🟡 56% (14/25) | | -| [context_window](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522context_window%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520context_window%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟡 90% (9/10) | 🟡 90% (9/10) | 🟡 50% (5/10) | 🟡 20% (2/10) | 🟡 90% (9/10) | | -| [counting](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522counting%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520counting%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (5/5) | 🟢 100% (5/5) | 🟢 100% (5/5) | 🟢 100% (5/5) | 🟢 100% (5/5) | | -| [datetime](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522datetime%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520datetime%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟡 87% (13/15) | 🟡 87% (13/15) | 🟡 67% (10/15) | 🟡 47% (7/15) | 🟡 93% (14/15) | | -| [easy](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522easy%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520easy%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟡 89% (31/35) | 🟡 63% (22/35) | 🟡 66% (23/35) | 🟡 86% (30/35) | 🟢 100% (35/35) | | -| [grafana-dashboard](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522grafana-dashboard%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520grafana-dashboard%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🔴 0% (0/5) | 🔴 0% (0/5) | 🔴 0% (0/5) | 🔴 0% (0/5) | 🔴 0% (0/5) | | -| [hard](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522hard%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520hard%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🔴 0% (0/5) | 🔴 0% (0/5) | 🔴 0% (0/5) | 🔴 0% (0/5) | 🔴 0% (0/5) | | -| [kubernetes](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522kubernetes%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520kubernetes%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟡 71% (25/35) | 🟡 43% (15/35) | 🟡 46% (16/35) | 🟡 63% (22/35) | 🟡 86% (30/35) | | -| [logs](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522logs%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520logs%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟡 60% (12/20) | 🟡 70% (14/20) | 🟡 50% (10/20) | 🟡 35% (7/20) | 🟡 70% (14/20) | | -| [loki](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522loki%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520loki%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟡 60% (3/5) | 🟢 100% (5/5) | 🟢 100% (5/5) | 🟡 80% (4/5) | 🟢 100% (5/5) | | -| [medium](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522medium%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520medium%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟡 60% (18/30) | 🟡 50% (15/30) | 🟡 43% (13/30) | 🟡 47% (14/30) | 🟡 80% (24/30) | | -| [metrics](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522metrics%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520metrics%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🔴 0% (0/5) | 🔴 0% (0/5) | 🔴 0% (0/5) | 🔴 0% (0/5) | 🔴 0% (0/5) | | -| [network](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522network%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520network%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (5/5) | 🟢 100% (5/5) | 🟡 80% (4/5) | 🟢 100% (5/5) | 🟢 100% (5/5) | | -| [one-test](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522one-test%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520one-test%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (5/5) | 🟡 40% (2/5) | 🟡 60% (3/5) | 🟢 100% (5/5) | 🟢 100% (5/5) | | -| [port-forward](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522port-forward%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520port-forward%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟡 30% (3/10) | 🟡 50% (5/10) | 🟡 50% (5/10) | 🟡 40% (4/10) | 🟡 50% (5/10) | | -| [question-answer](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522question-answer%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520question-answer%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🔴 0% (0/5) | 🔴 0% (0/5) | 🔴 0% (0/5) | 🔴 0% (0/5) | 🔴 0% (0/5) | | -| [regression](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522regression%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520regression%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟡 84% (38/45) | 🟡 56% (25/45) | 🟡 60% (27/45) | 🟡 82% (37/45) | 🟢 100% (45/45) | | -| [runbooks](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522runbooks%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520runbooks%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟡 40% (4/10) | 🟡 60% (6/10) | 🟡 60% (6/10) | 🟡 70% (7/10) | 🟢 100% (10/10) | | -| **Overall** | 🟡 70% (49/70) | 🟡 53% (37/70) | 🟡 51% (36/70) | 🟡 63% (44/70) | 🟡 84% (59/70) | | +| Tag | sonnet-4.5 | Warnings | +|-----|-------|----------| +| [easy](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522easy%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520easy%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | +| [kubernetes](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522kubernetes%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520kubernetes%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | +| [one-test](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522one-test%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520one-test%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | +| [regression](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522regression%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520regression%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | +| **Overall** | 🟢 100% (1/1) | | ## Raw Results @@ -78,42 +56,16 @@ Status of all evaluations across models. Color coding: - ⏱️ Timeout or rate limit error - ⏭️ Test skipped (e.g., known issue or precondition not met) -| Eval ID | [deepseek-3.1](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522deepseek-3.1%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520deepseek-3.1%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [gpt-5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522gpt-5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520gpt-5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [gpt-5.1](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522gpt-5.1%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520gpt-5.1%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [haiku-4.5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522haiku-4.5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520haiku-4.5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [sonnet-4.5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522sonnet-4.5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520sonnet-4.5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | -|---------|-------|-------|-------|-------|-------| -| [**09_crashpod**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | -| [**101_loki_historical_logs_pod_deleted**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/101_loki_historical_logs_pod_deleted/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%2522101_loki_historical_logs_pod_deleted%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%2520101_loki_historical_logs_pod_deleted%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522101_loki_historical_logs_pod_deleted%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520101_loki_historical_logs_pod_deleted%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522101_loki_historical_logs_pod_deleted%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520101_loki_historical_logs_pod_deleted%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522101_loki_historical_logs_pod_deleted%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520101_loki_historical_logs_pod_deleted%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522101_loki_historical_logs_pod_deleted%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520101_loki_historical_logs_pod_deleted%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522101_loki_historical_logs_pod_deleted%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520101_loki_historical_logs_pod_deleted%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | -| [**108_logs_nearby_lines**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/108_logs_nearby_lines/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%2522108_logs_nearby_lines%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%2520108_logs_nearby_lines%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522108_logs_nearby_lines%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520108_logs_nearby_lines%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522108_logs_nearby_lines%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520108_logs_nearby_lines%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522108_logs_nearby_lines%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520108_logs_nearby_lines%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522108_logs_nearby_lines%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520108_logs_nearby_lines%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522108_logs_nearby_lines%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520108_logs_nearby_lines%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | -| [**111_pod_names_contain_service**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/111_pod_names_contain_service/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%2522111_pod_names_contain_service%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%2520111_pod_names_contain_service%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522111_pod_names_contain_service%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520111_pod_names_contain_service%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522111_pod_names_contain_service%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520111_pod_names_contain_service%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522111_pod_names_contain_service%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520111_pod_names_contain_service%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522111_pod_names_contain_service%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520111_pod_names_contain_service%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522111_pod_names_contain_service%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520111_pod_names_contain_service%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | -| [**12_job_crashing**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/12_job_crashing/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252212_job_crashing%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252012_job_crashing%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252212_job_crashing%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252012_job_crashing%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252212_job_crashing%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252012_job_crashing%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252212_job_crashing%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252012_job_crashing%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252212_job_crashing%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252012_job_crashing%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252212_job_crashing%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252012_job_crashing%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | -| [**162_get_runbooks**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/162_get_runbooks/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%2522162_get_runbooks%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%2520162_get_runbooks%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522162_get_runbooks%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520162_get_runbooks%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522162_get_runbooks%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520162_get_runbooks%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522162_get_runbooks%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520162_get_runbooks%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522162_get_runbooks%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520162_get_runbooks%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522162_get_runbooks%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520162_get_runbooks%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | -| [**176_network_policy_blocking_traffic_no_runbooks**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/176_network_policy_blocking_traffic_no_runbooks/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%2522176_network_policy_blocking_traffic_no_runbooks%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%2520176_network_policy_blocking_traffic_no_runbooks%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522176_network_policy_blocking_traffic_no_runbooks%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520176_network_policy_blocking_traffic_no_runbooks%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522176_network_policy_blocking_traffic_no_runbooks%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520176_network_policy_blocking_traffic_no_runbooks%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522176_network_policy_blocking_traffic_no_runbooks%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520176_network_policy_blocking_traffic_no_runbooks%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522176_network_policy_blocking_traffic_no_runbooks%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520176_network_policy_blocking_traffic_no_runbooks%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522176_network_policy_blocking_traffic_no_runbooks%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520176_network_policy_blocking_traffic_no_runbooks%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | -| [**179_grafana_big_dashboard_query**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/179_grafana_big_dashboard_query/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%2522179_grafana_big_dashboard_query%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%2520179_grafana_big_dashboard_query%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522179_grafana_big_dashboard_query%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520179_grafana_big_dashboard_query%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522179_grafana_big_dashboard_query%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520179_grafana_big_dashboard_query%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522179_grafana_big_dashboard_query%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520179_grafana_big_dashboard_query%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522179_grafana_big_dashboard_query%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520179_grafana_big_dashboard_query%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522179_grafana_big_dashboard_query%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520179_grafana_big_dashboard_query%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | -| [**24_misconfigured_pvc**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/24_misconfigured_pvc/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252224_misconfigured_pvc%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252024_misconfigured_pvc%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252224_misconfigured_pvc%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252024_misconfigured_pvc%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252224_misconfigured_pvc%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252024_misconfigured_pvc%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252224_misconfigured_pvc%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252024_misconfigured_pvc%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252224_misconfigured_pvc%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252024_misconfigured_pvc%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252224_misconfigured_pvc%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252024_misconfigured_pvc%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | -| [**43_current_datetime_from_prompt**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/43_current_datetime_from_prompt/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252243_current_datetime_from_prompt%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252043_current_datetime_from_prompt%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252243_current_datetime_from_prompt%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252043_current_datetime_from_prompt%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252243_current_datetime_from_prompt%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252043_current_datetime_from_prompt%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252243_current_datetime_from_prompt%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252043_current_datetime_from_prompt%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252243_current_datetime_from_prompt%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252043_current_datetime_from_prompt%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252243_current_datetime_from_prompt%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252043_current_datetime_from_prompt%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | -| [**61_exact_match_counting**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/61_exact_match_counting/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252261_exact_match_counting%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252061_exact_match_counting%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252261_exact_match_counting%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252061_exact_match_counting%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252261_exact_match_counting%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252061_exact_match_counting%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252261_exact_match_counting%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252061_exact_match_counting%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252261_exact_match_counting%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252061_exact_match_counting%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252261_exact_match_counting%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252061_exact_match_counting%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | -| [**73a_time_window_anomaly**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/73a_time_window_anomaly/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252273a_time_window_anomaly%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252073a_time_window_anomaly%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273a_time_window_anomaly%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073a_time_window_anomaly%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273a_time_window_anomaly%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073a_time_window_anomaly%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273a_time_window_anomaly%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073a_time_window_anomaly%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273a_time_window_anomaly%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073a_time_window_anomaly%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273a_time_window_anomaly%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073a_time_window_anomaly%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | -| [**73b_time_window_anomaly**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/73b_time_window_anomaly/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252273b_time_window_anomaly%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252073b_time_window_anomaly%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273b_time_window_anomaly%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073b_time_window_anomaly%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273b_time_window_anomaly%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073b_time_window_anomaly%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273b_time_window_anomaly%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073b_time_window_anomaly%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273b_time_window_anomaly%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073b_time_window_anomaly%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273b_time_window_anomaly%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073b_time_window_anomaly%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | -| [**96_no_matching_runbook**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/96_no_matching_runbook/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252296_no_matching_runbook%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252096_no_matching_runbook%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252296_no_matching_runbook%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252096_no_matching_runbook%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252296_no_matching_runbook%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252096_no_matching_runbook%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252296_no_matching_runbook%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252096_no_matching_runbook%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252296_no_matching_runbook%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252096_no_matching_runbook%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252296_no_matching_runbook%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252096_no_matching_runbook%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | -| **SUMMARY** | 🟡 70% (49/70) | 🟡 53% (37/70) | 🟡 51% (36/70) | 🟡 63% (44/70) | 🟡 84% (59/70) | +| Eval ID | [sonnet-4.5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522sonnet-4.5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520sonnet-4.5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +|---------|-------| +| [**09_crashpod**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +| **SUMMARY** | 🟢 100% (1/1) | ## Detailed Raw Results -| Eval ID | deepseek-3.1 | gpt-5 | gpt-5.1 | haiku-4.5 | sonnet-4.5 | -|---------|-------|-------|-------|-------|-------| -| [09_crashpod](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 83.5s | [🟡 40% (2/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 15.1s / 💰 $0.04 | [🟡 60% (3/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 49.8s / 💰 $0.06 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 26.4s / 💰 $0.03 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 38.1s / 💰 $0.13 | -| [101_loki_historical_logs_pod_deleted](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/101_loki_historical_logs_pod_deleted/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%2522101_loki_historical_logs_pod_deleted%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%2520101_loki_historical_logs_pod_deleted%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡 60% (3/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522101_loki_historical_logs_pod_deleted%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520101_loki_historical_logs_pod_deleted%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 110.9s | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522101_loki_historical_logs_pod_deleted%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520101_loki_historical_logs_pod_deleted%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 29.9s / 💰 $0.08 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522101_loki_historical_logs_pod_deleted%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520101_loki_historical_logs_pod_deleted%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 196.8s / 💰 $0.24 | [🟡 80% (4/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522101_loki_historical_logs_pod_deleted%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520101_loki_historical_logs_pod_deleted%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 36.0s / 💰 $0.05 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522101_loki_historical_logs_pod_deleted%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520101_loki_historical_logs_pod_deleted%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 45.9s / 💰 $0.17 | -| [108_logs_nearby_lines](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/108_logs_nearby_lines/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%2522108_logs_nearby_lines%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%2520108_logs_nearby_lines%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴 0% (0/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522108_logs_nearby_lines%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520108_logs_nearby_lines%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 134.4s | [🔴 0% (0/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522108_logs_nearby_lines%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520108_logs_nearby_lines%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 33.2s / 💰 $0.09 | [🔴 0% (0/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522108_logs_nearby_lines%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520108_logs_nearby_lines%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 131.9s / 💰 $0.18 | [🟡 20% (1/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522108_logs_nearby_lines%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520108_logs_nearby_lines%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 36.5s / 💰 $0.06 | [🔴 0% (0/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522108_logs_nearby_lines%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520108_logs_nearby_lines%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 63.4s / 💰 $0.21 | -| [111_pod_names_contain_service](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/111_pod_names_contain_service/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%2522111_pod_names_contain_service%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%2520111_pod_names_contain_service%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522111_pod_names_contain_service%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520111_pod_names_contain_service%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 95.7s | [🔴 0% (0/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522111_pod_names_contain_service%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520111_pod_names_contain_service%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 12.4s / 💰 $0.02 | [🟡 40% (2/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522111_pod_names_contain_service%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520111_pod_names_contain_service%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 67.5s / 💰 $0.07 | [🟡 80% (4/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522111_pod_names_contain_service%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520111_pod_names_contain_service%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 24.0s / 💰 $0.03 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522111_pod_names_contain_service%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520111_pod_names_contain_service%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 53.5s / 💰 $0.23 | -| [12_job_crashing](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/12_job_crashing/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252212_job_crashing%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252012_job_crashing%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡 80% (4/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252212_job_crashing%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252012_job_crashing%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 95.0s | [🟡 20% (1/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252212_job_crashing%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252012_job_crashing%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 31.4s / 💰 $0.07 | [🟡 20% (1/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252212_job_crashing%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252012_job_crashing%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 85.4s / 💰 $0.11 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252212_job_crashing%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252012_job_crashing%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 31.0s / 💰 $0.04 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252212_job_crashing%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252012_job_crashing%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 48.8s / 💰 $0.16 | -| [162_get_runbooks](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/162_get_runbooks/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%2522162_get_runbooks%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%2520162_get_runbooks%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡 40% (2/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522162_get_runbooks%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520162_get_runbooks%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 82.2s | [🟡 60% (3/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522162_get_runbooks%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520162_get_runbooks%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 159.8s / 💰 $0.13 | [🟡 40% (2/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522162_get_runbooks%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520162_get_runbooks%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 96.1s / 💰 $0.12 | [🟡 60% (3/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522162_get_runbooks%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520162_get_runbooks%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 36.0s / 💰 $0.07 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522162_get_runbooks%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520162_get_runbooks%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 50.5s / 💰 $0.21 | -| [176_network_policy_blocking_traffic_no_runbooks](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/176_network_policy_blocking_traffic_no_runbooks/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%2522176_network_policy_blocking_traffic_no_runbooks%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%2520176_network_policy_blocking_traffic_no_runbooks%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522176_network_policy_blocking_traffic_no_runbooks%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520176_network_policy_blocking_traffic_no_runbooks%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 93.0s | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522176_network_policy_blocking_traffic_no_runbooks%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520176_network_policy_blocking_traffic_no_runbooks%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 35.3s / 💰 $0.09 | [🟡 80% (4/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522176_network_policy_blocking_traffic_no_runbooks%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520176_network_policy_blocking_traffic_no_runbooks%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 91.9s / 💰 $0.11 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522176_network_policy_blocking_traffic_no_runbooks%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520176_network_policy_blocking_traffic_no_runbooks%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 37.8s / 💰 $0.06 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522176_network_policy_blocking_traffic_no_runbooks%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520176_network_policy_blocking_traffic_no_runbooks%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 49.4s / 💰 $0.21 | -| [179_grafana_big_dashboard_query](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/179_grafana_big_dashboard_query/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%2522179_grafana_big_dashboard_query%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%2520179_grafana_big_dashboard_query%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴 0% (0/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522179_grafana_big_dashboard_query%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520179_grafana_big_dashboard_query%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 94.4s | [🔴 0% (0/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522179_grafana_big_dashboard_query%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520179_grafana_big_dashboard_query%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 21.2s / 💰 $0.04 | [🔴 0% (0/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522179_grafana_big_dashboard_query%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520179_grafana_big_dashboard_query%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 175.0s / 💰 $0.20 | [🔴 0% (0/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522179_grafana_big_dashboard_query%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520179_grafana_big_dashboard_query%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 29.5s / 💰 $0.04 | [🔴 0% (0/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%2522179_grafana_big_dashboard_query%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%2520179_grafana_big_dashboard_query%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 34.4s / 💰 $0.13 | -| [24_misconfigured_pvc](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/24_misconfigured_pvc/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252224_misconfigured_pvc%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252024_misconfigured_pvc%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252224_misconfigured_pvc%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252024_misconfigured_pvc%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 103.6s | [🔴 0% (0/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252224_misconfigured_pvc%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252024_misconfigured_pvc%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 4.3s / 💰 $0.00 | [🔴 0% (0/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252224_misconfigured_pvc%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252024_misconfigured_pvc%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 16.9s / 💰 $0.01 | [🟡 20% (1/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252224_misconfigured_pvc%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252024_misconfigured_pvc%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 9.9s / 💰 $0.01 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252224_misconfigured_pvc%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252024_misconfigured_pvc%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 42.5s / 💰 $0.14 | -| [43_current_datetime_from_prompt](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/43_current_datetime_from_prompt/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252243_current_datetime_from_prompt%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252043_current_datetime_from_prompt%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡 80% (4/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252243_current_datetime_from_prompt%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252043_current_datetime_from_prompt%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 12.9s | [🟡 80% (4/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252243_current_datetime_from_prompt%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252043_current_datetime_from_prompt%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 7.4s / 💰 $0.01 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252243_current_datetime_from_prompt%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252043_current_datetime_from_prompt%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 24.0s / 💰 $0.02 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252243_current_datetime_from_prompt%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252043_current_datetime_from_prompt%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 6.3s / 💰 $0.01 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252243_current_datetime_from_prompt%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252043_current_datetime_from_prompt%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 9.8s / 💰 $0.01 | -| [61_exact_match_counting](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/61_exact_match_counting/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252261_exact_match_counting%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252061_exact_match_counting%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252261_exact_match_counting%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252061_exact_match_counting%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 35.6s | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252261_exact_match_counting%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252061_exact_match_counting%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 16.4s / 💰 $0.04 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252261_exact_match_counting%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252061_exact_match_counting%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 49.8s / 💰 $0.06 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252261_exact_match_counting%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252061_exact_match_counting%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 14.0s / 💰 $0.01 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252261_exact_match_counting%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252061_exact_match_counting%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 19.1s / 💰 $0.06 | -| [73a_time_window_anomaly](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/73a_time_window_anomaly/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252273a_time_window_anomaly%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252073a_time_window_anomaly%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡 80% (4/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273a_time_window_anomaly%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073a_time_window_anomaly%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 95.6s | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273a_time_window_anomaly%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073a_time_window_anomaly%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 27.5s / 💰 $0.07 | [🟡 80% (4/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273a_time_window_anomaly%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073a_time_window_anomaly%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 93.0s / 💰 $0.13 | [🟡 20% (1/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273a_time_window_anomaly%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073a_time_window_anomaly%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 31.8s / 💰 $0.05 | [🟡 80% (4/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273a_time_window_anomaly%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073a_time_window_anomaly%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 48.6s / 💰 $0.20 | -| [73b_time_window_anomaly](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/73b_time_window_anomaly/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252273b_time_window_anomaly%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252073b_time_window_anomaly%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273b_time_window_anomaly%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073b_time_window_anomaly%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 93.2s | [🟡 80% (4/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273b_time_window_anomaly%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073b_time_window_anomaly%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 28.2s / 💰 $0.07 | [🟡 20% (1/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273b_time_window_anomaly%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073b_time_window_anomaly%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 37.3s / 💰 $0.03 | [🟡 20% (1/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273b_time_window_anomaly%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073b_time_window_anomaly%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 32.9s / 💰 $0.05 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252273b_time_window_anomaly%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252073b_time_window_anomaly%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 46.3s / 💰 $0.18 | -| [96_no_matching_runbook](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/96_no_matching_runbook/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252296_no_matching_runbook%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252096_no_matching_runbook%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟡 40% (2/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252296_no_matching_runbook%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252096_no_matching_runbook%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 99.9s | [🟡 60% (3/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252296_no_matching_runbook%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252096_no_matching_runbook%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 41.0s / 💰 $0.13 | [🟡 80% (4/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252296_no_matching_runbook%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252096_no_matching_runbook%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 132.4s / 💰 $0.17 | [🟡 80% (4/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252296_no_matching_runbook%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252096_no_matching_runbook%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 33.5s / 💰 $0.06 | [🟢 100% (5/5)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252296_no_matching_runbook%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252096_no_matching_runbook%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 60.6s / 💰 $0.30 | +| Eval ID | sonnet-4.5 | +|---------|-------| +| [09_crashpod](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 40.5s / 💰 $0.18 | --- -*Results are automatically generated and updated weekly. View full traces and detailed analysis in [Braintrust experiment: local-benchmark-20260101-140005](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260101-140005).* +*Results are automatically generated and updated weekly. View full traces and detailed analysis in [Braintrust experiment: local-benchmark-20260104-174426](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426).* diff --git a/run_benchmarks_local.py b/run_benchmarks_local.py index 9c6a5cd7d6..129a8b6879 100755 --- a/run_benchmarks_local.py +++ b/run_benchmarks_local.py @@ -8,12 +8,19 @@ import argparse import os +import shutil import subprocess import sys from datetime import datetime from pathlib import Path from typing import List, Optional +# Benchmark type definitions +BENCHMARK_TYPES = { + "fast-benchmark": "regression or benchmark", + "full-benchmark": "easy or medium or hard or regression or benchmark", +} + class BenchmarkRunner: """Manages local benchmark execution for HolmesGPT evaluations.""" @@ -27,6 +34,7 @@ def __init__( parallel_workers: Optional[str] = "auto", strict_setup: bool = True, no_braintrust: bool = False, + benchmark_type: Optional[str] = None, ): self.models = models self.markers = markers @@ -35,6 +43,7 @@ def __init__( self.parallel_workers = parallel_workers self.strict_setup = strict_setup self.no_braintrust = no_braintrust + self.benchmark_type = benchmark_type self.experiment_id = os.environ.get( "EXPERIMENT_ID", f"local-benchmark-{datetime.now().strftime('%Y%m%d-%H%M%S')}", @@ -45,6 +54,8 @@ def check_environment(self) -> None: print("=" * 50) print("🧪 Running Local Benchmarks") print("=" * 50) + if self.benchmark_type: + print(f"Benchmark: {self.benchmark_type}") print(f"Models: {', '.join(self.models)}") print(f"Markers: llm and ({self.markers})") print(f"Iterations: {self.iterations}") @@ -196,13 +207,25 @@ def generate_report(self) -> None: ) return - # Create output directories + # Create output directories based on benchmark type docs_dir = Path("docs/development/evaluations") - history_dir = docs_dir / "history" + + # Determine output file and history directory based on benchmark type + if self.benchmark_type == "fast-benchmark": + main_output = docs_dir / "fast-benchmark-results.md" + history_subdir = "fast" + elif self.benchmark_type == "full-benchmark": + main_output = docs_dir / "full-benchmark-results.md" + history_subdir = "full" + else: + # Custom markers - use full benchmark output location + main_output = docs_dir / "full-benchmark-results.md" + history_subdir = "full" + + history_dir = docs_dir / "history" / history_subdir history_dir.mkdir(parents=True, exist_ok=True) - # Generate latest results - latest_output = docs_dir / "latest-results.md" + # Build base command cmd = [ "poetry", "run", @@ -211,21 +234,34 @@ def generate_report(self) -> None: "--json-file", "eval_results.json", "--output-file", - str(latest_output), + str(main_output), "--models", ",".join(self.models), ] + # Add benchmark type if specified + if self.benchmark_type: + cmd.extend(["--benchmark-type", self.benchmark_type]) + try: subprocess.run(cmd, check=True) - print(f"✅ Report generated: {latest_output}") + print(f"✅ Report generated: {main_output}") + + # For fast-benchmark, also copy to latest-results.md for backwards compatibility + if self.benchmark_type == "fast-benchmark": + latest_output = docs_dir / "latest-results.md" + shutil.copy(main_output, latest_output) + print(f"📋 Copied to: {latest_output} (backwards compatibility)") # Generate historical copy timestamp = datetime.now().strftime("%Y%m%d_%H%M%S") history_output = history_dir / f"results_{timestamp}.md" - cmd[-2] = str(history_output) # Update output file - subprocess.run(cmd, check=True) + # Update command for historical output + cmd_history = cmd.copy() + output_idx = cmd_history.index("--output-file") + 1 + cmd_history[output_idx] = str(history_output) + subprocess.run(cmd_history, check=True) print(f"📁 Saved historical copy: {history_output}") except subprocess.CalledProcessError as e: @@ -237,6 +273,8 @@ def show_summary(self) -> None: print("=" * 50) print("Test Execution Summary") print("=" * 50) + if self.benchmark_type: + print(f"Benchmark: {self.benchmark_type}") print(f"Models: {', '.join(self.models)}") print(f"Markers: llm and ({self.markers})") print(f"Iterations: {self.iterations}") @@ -254,12 +292,27 @@ def show_summary(self) -> None: print() print("Generated files:") + # Determine which result file to check based on benchmark type + if self.benchmark_type == "fast-benchmark": + result_file = "docs/development/evaluations/fast-benchmark-results.md" + else: + result_file = "docs/development/evaluations/full-benchmark-results.md" + files_to_check = [ ("eval_results.json", "JSON results"), ("evals_report.md", "Evaluation report"), - ("docs/development/evaluations/latest-results.md", "Latest results"), + (result_file, "Benchmark results"), ] + # For fast-benchmark, also check latest-results.md + if self.benchmark_type == "fast-benchmark": + files_to_check.append( + ( + "docs/development/evaluations/latest-results.md", + "Latest results (copy)", + ) + ) + for filename, description in files_to_check: path = Path(filename) if path.exists(): @@ -271,7 +324,9 @@ def show_summary(self) -> None: print("✅ Benchmark run complete!") print() print("To commit results (like CI/CD would on main):") - print(" git add docs/development/evaluations/latest-results.md") + print(f" git add {result_file}") + if self.benchmark_type == "fast-benchmark": + print(" git add docs/development/evaluations/latest-results.md") print(" git commit -m 'Update benchmark results [skip ci]'") print("=" * 50) @@ -291,10 +346,16 @@ def parse_args(): description="Run HolmesGPT evaluation benchmarks locally", formatter_class=argparse.RawDescriptionHelpFormatter, epilog=""" +Benchmark Types: + fast-benchmark - Quick regression tests (markers: regression or benchmark) + full-benchmark - Comprehensive tests (markers: easy or medium or hard or regression or benchmark) + Examples: - %(prog)s # Use all defaults - %(prog)s --models gpt-4o # Test only gpt-4o - %(prog)s --models gpt-4o,claude-3-5-sonnet --markers easy --iterations 3 + %(prog)s # Default: fast-benchmark with default models + %(prog)s --benchmark-type fast-benchmark # Explicit fast benchmark + %(prog)s --benchmark-type full-benchmark # Full comprehensive benchmark + %(prog)s --models gpt-4o # Test only gpt-4o (fast-benchmark) + %(prog)s --markers easy --iterations 3 # Custom markers (cannot combine with --benchmark-type) %(prog)s --filter 01_how_many_pods # Run specific test %(prog)s --parallel 6 # Run with 6 parallel workers %(prog)s --parallel 1 # Run sequentially (no parallelism) @@ -317,11 +378,19 @@ def parse_args(): help="Comma-separated list of models to test (default: %(default)s)", ) + parser.add_argument( + "--benchmark-type", + type=str, + choices=list(BENCHMARK_TYPES.keys()), + default=None, + help="Type of benchmark to run (default: fast-benchmark). Cannot be combined with --markers.", + ) + parser.add_argument( "--markers", type=str, - default="regression or benchmark", - help="Pytest markers for test selection (combined with 'llm') (default: %(default)s)", + default=None, + help="Custom pytest markers for test selection (combined with 'llm'). Cannot be combined with --benchmark-type.", ) parser.add_argument( @@ -366,17 +435,38 @@ def main(): """Main entry point.""" args = parse_args() + # Validate that --benchmark-type and --markers are not both provided + if args.markers is not None and args.benchmark_type is not None: + print("❌ ERROR: Cannot combine --benchmark-type with --markers") + print(" Use either --benchmark-type OR --markers, not both.") + sys.exit(1) + + # Determine markers and benchmark_type + if args.markers is not None: + # Custom markers provided - no benchmark type + markers = args.markers + benchmark_type = None + elif args.benchmark_type is not None: + # Explicit benchmark type + benchmark_type = args.benchmark_type + markers = BENCHMARK_TYPES[benchmark_type] + else: + # Default: fast-benchmark + benchmark_type = "fast-benchmark" + markers = BENCHMARK_TYPES[benchmark_type] + # Parse models from comma-separated string models = [m.strip() for m in args.models.split(",") if m.strip()] runner = BenchmarkRunner( models=models, - markers=args.markers, + markers=markers, iterations=args.iterations, filter_tests=args.filter_tests, parallel_workers=args.parallel_workers, strict_setup=args.strict_setup, no_braintrust=args.no_braintrust, + benchmark_type=benchmark_type, ) # Ignore the exit code from tests to match bash script behavior diff --git a/tests/generate_eval_report.py b/tests/generate_eval_report.py index 7b66457327..ab3ab17762 100755 --- a/tests/generate_eval_report.py +++ b/tests/generate_eval_report.py @@ -229,6 +229,12 @@ def parse_args(): "--models", help="Comma-separated list of models tested (auto-detected if not provided)", ) + parser.add_argument( + "--benchmark-type", + choices=["fast-benchmark", "full-benchmark"], + default=None, + help="Type of benchmark (fast-benchmark or full-benchmark)", + ) return parser.parse_args() @@ -1324,6 +1330,15 @@ def main(): # Generate report sections report_lines = [] + # Determine benchmark type label for title + benchmark_type = getattr(args, "benchmark_type", None) + if benchmark_type == "fast-benchmark": + benchmark_label = "Fast Benchmark" + elif benchmark_type == "full-benchmark": + benchmark_label = "Full Benchmark" + else: + benchmark_label = "Benchmark" + # Header - check if output file is in history format output_filename = Path(args.output_file).name import re @@ -1339,8 +1354,8 @@ def main(): title = date_obj.strftime("%B %d, %Y") report_lines.append(f"# {title}") else: - # Default title for latest-results.md - report_lines.append("# HolmesGPT LLM Evaluation Benchmark Results") + # Default title with benchmark type + report_lines.append(f"# HolmesGPT LLM Evaluation {benchmark_label} Results") report_lines.append("") # Format duration nicely duration_seconds = results.get("duration", 0) @@ -1419,9 +1434,26 @@ def main(): report_lines.append(f"**Judge (classifier) model**: {classifier_model}") report_lines.append("") - # About this benchmark - report_lines.append("## About this Benchmark") - report_lines.append("") + # About this benchmark - add info box for benchmark type + if benchmark_type == "fast-benchmark": + report_lines.append('!!! info "Fast Benchmark"') + report_lines.append(" **Markers**: `regression or benchmark`
") + report_lines.append(" **Schedule**: Weekly (Sunday 2 AM UTC)
") + report_lines.append( + " **Purpose**: Quick regression tests to catch breaking changes" + ) + report_lines.append("") + elif benchmark_type == "full-benchmark": + report_lines.append('!!! info "Full Benchmark"') + report_lines.append( + " **Markers**: `easy or medium or hard or regression or benchmark`
" + ) + report_lines.append(" **Schedule**: Manual / On-demand
") + report_lines.append( + " **Purpose**: Comprehensive testing across all difficulty levels" + ) + report_lines.append("") + report_lines.append( "HolmesGPT is continuously evaluated against real-world " "Kubernetes and cloud troubleshooting scenarios." From 4769cf9044adc457f80737d42c362414c2a93c08 Mon Sep 17 00:00:00 2001 From: Tomer Keshet Date: Mon, 5 Jan 2026 15:57:27 +0200 Subject: [PATCH 2/7] save fast benchmark next to regular benchmarks Signed-off-by: Tomer Keshet --- docs/development/evaluations/.nav.yml | 3 +- .../evaluations/fast-benchmark-results.md | 51 +++++++----- .../evaluations/history/{fast => }/.nav.yml | 0 .../custom_claude_results_20250930_153753.md | 0 ...tom_self_hosted_results_20251008_053744.md | 0 .../evaluations/history/fast/.gitkeep | 0 .../evaluations/history/fast/index.md | 12 --- .../history/fast/results_20260104_174350.md | 22 ----- .../history/fast/results_20260104_174620.md | 70 ---------------- .../evaluations/history/full/.gitkeep | 0 .../evaluations/history/full/.nav.yml | 5 -- .../evaluations/history/full/index.md | 16 ---- docs/development/evaluations/history/index.md | 10 +++ .../{full => }/results_20250928_001434.md | 0 .../{full => }/results_20250930_085923.md | 0 .../{full => }/results_20251012_170303.md | 0 .../{full => }/results_20251127_042958.md | 0 .../{fast => }/results_20260104_174301.md | 2 +- .../history/results_20260105_153903.md | 82 +++++++++++++++++++ docs/development/evaluations/index.md | 9 +- .../development/evaluations/latest-results.md | 51 +++++++----- run_benchmarks_local.py | 35 +++----- tests/generate_eval_report.py | 22 +++-- 23 files changed, 187 insertions(+), 203 deletions(-) rename docs/development/evaluations/history/{fast => }/.nav.yml (100%) rename docs/development/evaluations/history/{full => }/custom_claude_results_20250930_153753.md (100%) rename docs/development/evaluations/history/{full => }/custom_self_hosted_results_20251008_053744.md (100%) delete mode 100644 docs/development/evaluations/history/fast/.gitkeep delete mode 100644 docs/development/evaluations/history/fast/index.md delete mode 100644 docs/development/evaluations/history/fast/results_20260104_174350.md delete mode 100644 docs/development/evaluations/history/fast/results_20260104_174620.md delete mode 100644 docs/development/evaluations/history/full/.gitkeep delete mode 100644 docs/development/evaluations/history/full/.nav.yml delete mode 100644 docs/development/evaluations/history/full/index.md create mode 100644 docs/development/evaluations/history/index.md rename docs/development/evaluations/history/{full => }/results_20250928_001434.md (100%) rename docs/development/evaluations/history/{full => }/results_20250930_085923.md (100%) rename docs/development/evaluations/history/{full => }/results_20251012_170303.md (100%) rename docs/development/evaluations/history/{full => }/results_20251127_042958.md (100%) rename docs/development/evaluations/history/{fast => }/results_20260104_174301.md (99%) create mode 100644 docs/development/evaluations/history/results_20260105_153903.md diff --git a/docs/development/evaluations/.nav.yml b/docs/development/evaluations/.nav.yml index aeaeaa7145..0222eba3fb 100644 --- a/docs/development/evaluations/.nav.yml +++ b/docs/development/evaluations/.nav.yml @@ -1,8 +1,7 @@ nav: - index.md - Latest Results: latest-results.md - - Fast Benchmark: history/fast - - Full Benchmark: history/full + - History: history - Running Evaluations: running-evals.md - Adding New Evaluations: adding-evals.md - Benchmarking New Models: benchmarking-new-models.md diff --git a/docs/development/evaluations/fast-benchmark-results.md b/docs/development/evaluations/fast-benchmark-results.md index 39f863ff4a..d874fd13a4 100644 --- a/docs/development/evaluations/fast-benchmark-results.md +++ b/docs/development/evaluations/fast-benchmark-results.md @@ -1,7 +1,7 @@ -# HolmesGPT LLM Evaluation Fast Benchmark Results +# ⚡ HolmesGPT LLM Evaluation Fast Benchmark Results -**Generated**: 2026-01-04 18:10 UTC -**Total Duration**: 1m 18s +**Generated**: 2026-01-05 15:39 UTC +**Total Duration**: 3m 4s **Iterations**: 1 **Judge (classifier) model**: gpt-4.1 @@ -18,31 +18,42 @@ If you find scenarios that HolmesGPT does not perform well on, please consider a | Model | Pass | Fail | Skip/Error | Total | Success Rate | |-------|------|------|------------|-------|--------------| +| deepseek-3.1 | 1 | 0 | 0 | 1 | 🟢 100% (1/1) | +| gpt-5 | 0 | 1 | 0 | 1 | 🔴 0% (0/1) | +| gpt-5.1 | 1 | 0 | 0 | 1 | 🟢 100% (1/1) | +| haiku-4.5 | 1 | 0 | 0 | 1 | 🟢 100% (1/1) | | sonnet-4.5 | 1 | 0 | 0 | 1 | 🟢 100% (1/1) | ## Model Cost Comparison | Model | Tests | Avg Cost | Min Cost | Max Cost | Total Cost | |-------|-------|----------|----------|----------|------------| -| sonnet-4.5 | 1 | $0.18 | $0.18 | $0.18 | $0.18 | +| gpt-5 | 1 | $0.03 | $0.03 | $0.03 | $0.03 | +| gpt-5.1 | 1 | $0.12 | $0.12 | $0.12 | $0.12 | +| haiku-4.5 | 1 | $0.05 | $0.05 | $0.05 | $0.05 | +| sonnet-4.5 | 1 | $0.19 | $0.19 | $0.19 | $0.19 | ## Model Latency Comparison | Model | Avg (s) | Min (s) | Max (s) | P50 (s) | P95 (s) | |-------|---------|---------|---------|---------|---------| -| sonnet-4.5 | 40.5 | 40.5 | 40.5 | 40.5 | 40.5 | +| deepseek-3.1 | 131.8 | 131.8 | 131.8 | 131.8 | 131.8 | +| gpt-5 | 23.4 | 23.4 | 23.4 | 23.4 | 23.4 | +| gpt-5.1 | 73.1 | 73.1 | 73.1 | 73.1 | 73.1 | +| haiku-4.5 | 31.0 | 31.0 | 31.0 | 31.0 | 31.0 | +| sonnet-4.5 | 47.8 | 47.8 | 47.8 | 47.8 | 47.8 | ## Performance by Tag Success rate by test category and model: -| Tag | sonnet-4.5 | Warnings | -|-----|-------|----------| -| [easy](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522easy%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520easy%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | -| [kubernetes](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522kubernetes%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520kubernetes%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | -| [one-test](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522one-test%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520one-test%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | -| [regression](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522regression%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520regression%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | -| **Overall** | 🟢 100% (1/1) | | +| Tag | deepseek-3.1 | gpt-5 | gpt-5.1 | haiku-4.5 | sonnet-4.5 | Warnings | +|-----|-------|-------|-------|-------|-------|----------| +| [easy](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522easy%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520easy%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | | +| [kubernetes](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522kubernetes%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520kubernetes%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | | +| [one-test](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522one-test%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520one-test%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | | +| [regression](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522regression%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520regression%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | | +| **Overall** | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | | ## Raw Results @@ -56,16 +67,16 @@ Status of all evaluations across models. Color coding: - ⏱️ Timeout or rate limit error - ⏭️ Test skipped (e.g., known issue or precondition not met) -| Eval ID | [sonnet-4.5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522sonnet-4.5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520sonnet-4.5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | -|---------|-------| -| [**09_crashpod**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | -| **SUMMARY** | 🟢 100% (1/1) | +| Eval ID | [deepseek-3.1](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522deepseek-3.1%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520deepseek-3.1%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [gpt-5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522gpt-5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520gpt-5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [gpt-5.1](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522gpt-5.1%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520gpt-5.1%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [haiku-4.5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522haiku-4.5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520haiku-4.5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [sonnet-4.5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522sonnet-4.5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520sonnet-4.5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +|---------|-------|-------|-------|-------|-------| +| [**09_crashpod**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +| **SUMMARY** | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | ## Detailed Raw Results -| Eval ID | sonnet-4.5 | -|---------|-------| -| [09_crashpod](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 40.5s / 💰 $0.18 | +| Eval ID | deepseek-3.1 | gpt-5 | gpt-5.1 | haiku-4.5 | sonnet-4.5 | +|---------|-------|-------|-------|-------|-------| +| [09_crashpod](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 131.8s | [🔴 0% (0/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 23.4s / 💰 $0.03 | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 73.1s / 💰 $0.12 | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 31.0s / 💰 $0.05 | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 47.8s / 💰 $0.19 | --- -*Results are automatically generated and updated weekly. View full traces and detailed analysis in [Braintrust experiment: local-benchmark-20260104-174426](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426).* +*Results are automatically generated and updated weekly. View full traces and detailed analysis in [Braintrust experiment: local-benchmark-20260105-153457](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457).* diff --git a/docs/development/evaluations/history/fast/.nav.yml b/docs/development/evaluations/history/.nav.yml similarity index 100% rename from docs/development/evaluations/history/fast/.nav.yml rename to docs/development/evaluations/history/.nav.yml diff --git a/docs/development/evaluations/history/full/custom_claude_results_20250930_153753.md b/docs/development/evaluations/history/custom_claude_results_20250930_153753.md similarity index 100% rename from docs/development/evaluations/history/full/custom_claude_results_20250930_153753.md rename to docs/development/evaluations/history/custom_claude_results_20250930_153753.md diff --git a/docs/development/evaluations/history/full/custom_self_hosted_results_20251008_053744.md b/docs/development/evaluations/history/custom_self_hosted_results_20251008_053744.md similarity index 100% rename from docs/development/evaluations/history/full/custom_self_hosted_results_20251008_053744.md rename to docs/development/evaluations/history/custom_self_hosted_results_20251008_053744.md diff --git a/docs/development/evaluations/history/fast/.gitkeep b/docs/development/evaluations/history/fast/.gitkeep deleted file mode 100644 index e69de29bb2..0000000000 diff --git a/docs/development/evaluations/history/fast/index.md b/docs/development/evaluations/history/fast/index.md deleted file mode 100644 index 2d54f2f083..0000000000 --- a/docs/development/evaluations/history/fast/index.md +++ /dev/null @@ -1,12 +0,0 @@ -# Fast Benchmark Results - -!!! info "Fast Benchmark" - **Markers**: `regression or benchmark`
- **Schedule**: Weekly (Sunday 2 AM UTC)
- **Purpose**: Quick regression tests to catch breaking changes - -Fast benchmarks run a subset of critical tests designed to complete quickly while catching regressions. These are the primary automated benchmarks that run weekly. - -## Recent Results - -Browse the results below to track fast benchmark performance over time. diff --git a/docs/development/evaluations/history/fast/results_20260104_174350.md b/docs/development/evaluations/history/fast/results_20260104_174350.md deleted file mode 100644 index 6237ae47eb..0000000000 --- a/docs/development/evaluations/history/fast/results_20260104_174350.md +++ /dev/null @@ -1,22 +0,0 @@ -# January 04, 2026 - -**Generated**: 2026-01-04 17:43 UTC -**Total Duration**: 10s -**Iterations**: 1 -**Judge (classifier) model**: gpt-4.1 - -## About this Benchmark - -**Fast Benchmark**: Quick regression tests using markers `regression or benchmark` - designed to run frequently and catch regressions. - -HolmesGPT is continuously evaluated against real-world Kubernetes and cloud troubleshooting scenarios. - -If you find scenarios that HolmesGPT does not perform well on, please consider adding them as evals to the benchmark. - -## Model Accuracy Comparison - -| Model | Pass | Fail | Skip/Error | Total | Success Rate | -|-------|------|------|------------|-------|--------------| - ---- -*Results are automatically generated and updated weekly. For detailed traces and analysis, see our [Braintrust dashboard](https://braintrust.dev).* diff --git a/docs/development/evaluations/history/fast/results_20260104_174620.md b/docs/development/evaluations/history/fast/results_20260104_174620.md deleted file mode 100644 index 6285c1a4f0..0000000000 --- a/docs/development/evaluations/history/fast/results_20260104_174620.md +++ /dev/null @@ -1,70 +0,0 @@ -# January 04, 2026 - -**Generated**: 2026-01-04 17:46 UTC -**Total Duration**: 1m 18s -**Iterations**: 1 -**Judge (classifier) model**: gpt-4.1 - -## About this Benchmark - -**Fast Benchmark**: Quick regression tests using markers `regression or benchmark` - designed to run frequently and catch regressions. - -HolmesGPT is continuously evaluated against real-world Kubernetes and cloud troubleshooting scenarios. - -If you find scenarios that HolmesGPT does not perform well on, please consider adding them as evals to the benchmark. - -## Model Accuracy Comparison - -| Model | Pass | Fail | Skip/Error | Total | Success Rate | -|-------|------|------|------------|-------|--------------| -| sonnet-4.5 | 1 | 0 | 0 | 1 | 🟢 100% (1/1) | - -## Model Cost Comparison - -| Model | Tests | Avg Cost | Min Cost | Max Cost | Total Cost | -|-------|-------|----------|----------|----------|------------| -| sonnet-4.5 | 1 | $0.18 | $0.18 | $0.18 | $0.18 | - -## Model Latency Comparison - -| Model | Avg (s) | Min (s) | Max (s) | P50 (s) | P95 (s) | -|-------|---------|---------|---------|---------|---------| -| sonnet-4.5 | 40.5 | 40.5 | 40.5 | 40.5 | 40.5 | - -## Performance by Tag - -Success rate by test category and model: - -| Tag | sonnet-4.5 | Warnings | -|-----|-------|----------| -| [easy](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522easy%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520easy%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | -| [kubernetes](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522kubernetes%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520kubernetes%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | -| [one-test](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522one-test%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520one-test%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | -| [regression](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522regression%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520regression%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | -| **Overall** | 🟢 100% (1/1) | | - -## Raw Results - -Status of all evaluations across models. Color coding: - -- 🟢 Passing 100% (stable) -- 🟡 Passing 1-99% -- 🔴 Passing 0% (failing) -- 🔧 Mock data failure (missing or invalid test data) -- ⚠️ Setup failure (environment/infrastructure issue) -- ⏱️ Timeout or rate limit error -- ⏭️ Test skipped (e.g., known issue or precondition not met) - -| Eval ID | [sonnet-4.5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522sonnet-4.5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520sonnet-4.5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | -|---------|-------| -| [**09_crashpod**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | -| **SUMMARY** | 🟢 100% (1/1) | - -## Detailed Raw Results - -| Eval ID | sonnet-4.5 | -|---------|-------| -| [09_crashpod](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 40.5s / 💰 $0.18 | - ---- -*Results are automatically generated and updated weekly. View full traces and detailed analysis in [Braintrust experiment: local-benchmark-20260104-174426](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426).* diff --git a/docs/development/evaluations/history/full/.gitkeep b/docs/development/evaluations/history/full/.gitkeep deleted file mode 100644 index e69de29bb2..0000000000 diff --git a/docs/development/evaluations/history/full/.nav.yml b/docs/development/evaluations/history/full/.nav.yml deleted file mode 100644 index d71105aebd..0000000000 --- a/docs/development/evaluations/history/full/.nav.yml +++ /dev/null @@ -1,5 +0,0 @@ -sort: - direction: desc -nav: - - index.md - - "*" diff --git a/docs/development/evaluations/history/full/index.md b/docs/development/evaluations/history/full/index.md deleted file mode 100644 index 4d284a7722..0000000000 --- a/docs/development/evaluations/history/full/index.md +++ /dev/null @@ -1,16 +0,0 @@ -# Full Benchmark Results - -!!! info "Full Benchmark" - **Markers**: `easy or medium or hard or regression or benchmark`
- **Schedule**: Manual / On-demand
- **Purpose**: Comprehensive testing across all difficulty levels - -Full benchmarks run the complete test suite including easy, medium, and hard difficulty tests. These provide comprehensive coverage but take longer to complete. - -## Historical Results - -This section includes: - -- **Full benchmark runs** - Comprehensive test suite executions -- **Extended comparisons** - Special runs comparing multiple models -- **Legacy results** - Historical benchmark data from before the fast/full separation diff --git a/docs/development/evaluations/history/index.md b/docs/development/evaluations/history/index.md new file mode 100644 index 0000000000..137f42b95d --- /dev/null +++ b/docs/development/evaluations/history/index.md @@ -0,0 +1,10 @@ +# Benchmark History + +Browse through past benchmark runs to track performance trends over time. + +Results marked with ⚡ are **fast benchmarks** (quick regression tests). Other results are **full benchmarks** (comprehensive test suites). + +| Type | Markers | Purpose | +|------|---------|---------| +| ⚡ Fast | `regression or benchmark` | Quick regression tests | +| Full | `easy or medium or hard or regression or benchmark` | Comprehensive testing | diff --git a/docs/development/evaluations/history/full/results_20250928_001434.md b/docs/development/evaluations/history/results_20250928_001434.md similarity index 100% rename from docs/development/evaluations/history/full/results_20250928_001434.md rename to docs/development/evaluations/history/results_20250928_001434.md diff --git a/docs/development/evaluations/history/full/results_20250930_085923.md b/docs/development/evaluations/history/results_20250930_085923.md similarity index 100% rename from docs/development/evaluations/history/full/results_20250930_085923.md rename to docs/development/evaluations/history/results_20250930_085923.md diff --git a/docs/development/evaluations/history/full/results_20251012_170303.md b/docs/development/evaluations/history/results_20251012_170303.md similarity index 100% rename from docs/development/evaluations/history/full/results_20251012_170303.md rename to docs/development/evaluations/history/results_20251012_170303.md diff --git a/docs/development/evaluations/history/full/results_20251127_042958.md b/docs/development/evaluations/history/results_20251127_042958.md similarity index 100% rename from docs/development/evaluations/history/full/results_20251127_042958.md rename to docs/development/evaluations/history/results_20251127_042958.md diff --git a/docs/development/evaluations/history/fast/results_20260104_174301.md b/docs/development/evaluations/history/results_20260104_174301.md similarity index 99% rename from docs/development/evaluations/history/fast/results_20260104_174301.md rename to docs/development/evaluations/history/results_20260104_174301.md index db38c0879c..48ad60a2e3 100644 --- a/docs/development/evaluations/history/fast/results_20260104_174301.md +++ b/docs/development/evaluations/history/results_20260104_174301.md @@ -1,4 +1,4 @@ -# January 04, 2026 +# ⚡ January 04, 2026 **Generated**: 2026-01-04 17:43 UTC **Total Duration**: 43m 0s diff --git a/docs/development/evaluations/history/results_20260105_153903.md b/docs/development/evaluations/history/results_20260105_153903.md new file mode 100644 index 0000000000..b7ea324c64 --- /dev/null +++ b/docs/development/evaluations/history/results_20260105_153903.md @@ -0,0 +1,82 @@ +# ⚡ January 05, 2026 + +**Generated**: 2026-01-05 15:39 UTC +**Total Duration**: 3m 4s +**Iterations**: 1 +**Judge (classifier) model**: gpt-4.1 + +!!! info "Fast Benchmark" + **Markers**: `regression or benchmark`
+ **Schedule**: Weekly (Sunday 2 AM UTC)
+ **Purpose**: Quick regression tests to catch breaking changes + +HolmesGPT is continuously evaluated against real-world Kubernetes and cloud troubleshooting scenarios. + +If you find scenarios that HolmesGPT does not perform well on, please consider adding them as evals to the benchmark. + +## Model Accuracy Comparison + +| Model | Pass | Fail | Skip/Error | Total | Success Rate | +|-------|------|------|------------|-------|--------------| +| deepseek-3.1 | 1 | 0 | 0 | 1 | 🟢 100% (1/1) | +| gpt-5 | 0 | 1 | 0 | 1 | 🔴 0% (0/1) | +| gpt-5.1 | 1 | 0 | 0 | 1 | 🟢 100% (1/1) | +| haiku-4.5 | 1 | 0 | 0 | 1 | 🟢 100% (1/1) | +| sonnet-4.5 | 1 | 0 | 0 | 1 | 🟢 100% (1/1) | + +## Model Cost Comparison + +| Model | Tests | Avg Cost | Min Cost | Max Cost | Total Cost | +|-------|-------|----------|----------|----------|------------| +| gpt-5 | 1 | $0.03 | $0.03 | $0.03 | $0.03 | +| gpt-5.1 | 1 | $0.12 | $0.12 | $0.12 | $0.12 | +| haiku-4.5 | 1 | $0.05 | $0.05 | $0.05 | $0.05 | +| sonnet-4.5 | 1 | $0.19 | $0.19 | $0.19 | $0.19 | + +## Model Latency Comparison + +| Model | Avg (s) | Min (s) | Max (s) | P50 (s) | P95 (s) | +|-------|---------|---------|---------|---------|---------| +| deepseek-3.1 | 131.8 | 131.8 | 131.8 | 131.8 | 131.8 | +| gpt-5 | 23.4 | 23.4 | 23.4 | 23.4 | 23.4 | +| gpt-5.1 | 73.1 | 73.1 | 73.1 | 73.1 | 73.1 | +| haiku-4.5 | 31.0 | 31.0 | 31.0 | 31.0 | 31.0 | +| sonnet-4.5 | 47.8 | 47.8 | 47.8 | 47.8 | 47.8 | + +## Performance by Tag + +Success rate by test category and model: + +| Tag | deepseek-3.1 | gpt-5 | gpt-5.1 | haiku-4.5 | sonnet-4.5 | Warnings | +|-----|-------|-------|-------|-------|-------|----------| +| [easy](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522easy%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520easy%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | | +| [kubernetes](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522kubernetes%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520kubernetes%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | | +| [one-test](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522one-test%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520one-test%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | | +| [regression](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522regression%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520regression%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | | +| **Overall** | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | | + +## Raw Results + +Status of all evaluations across models. Color coding: + +- 🟢 Passing 100% (stable) +- 🟡 Passing 1-99% +- 🔴 Passing 0% (failing) +- 🔧 Mock data failure (missing or invalid test data) +- ⚠️ Setup failure (environment/infrastructure issue) +- ⏱️ Timeout or rate limit error +- ⏭️ Test skipped (e.g., known issue or precondition not met) + +| Eval ID | [deepseek-3.1](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522deepseek-3.1%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520deepseek-3.1%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [gpt-5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522gpt-5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520gpt-5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [gpt-5.1](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522gpt-5.1%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520gpt-5.1%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [haiku-4.5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522haiku-4.5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520haiku-4.5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [sonnet-4.5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522sonnet-4.5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520sonnet-4.5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +|---------|-------|-------|-------|-------|-------| +| [**09_crashpod**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +| **SUMMARY** | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | + +## Detailed Raw Results + +| Eval ID | deepseek-3.1 | gpt-5 | gpt-5.1 | haiku-4.5 | sonnet-4.5 | +|---------|-------|-------|-------|-------|-------| +| [09_crashpod](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 131.8s | [🔴 0% (0/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 23.4s / 💰 $0.03 | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 73.1s / 💰 $0.12 | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 31.0s / 💰 $0.05 | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 47.8s / 💰 $0.19 | + +--- +*Results are automatically generated and updated weekly. View full traces and detailed analysis in [Braintrust experiment: local-benchmark-20260105-153457](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457).* diff --git a/docs/development/evaluations/index.md b/docs/development/evaluations/index.md index 9d1c8ac1e6..4c497ab7b4 100644 --- a/docs/development/evaluations/index.md +++ b/docs/development/evaluations/index.md @@ -12,12 +12,15 @@ We run two types of benchmarks to balance speed and coverage: | Benchmark | Markers | Purpose | Schedule | |-----------|---------|---------|----------| -| **[Fast Benchmark](./history/fast/index.md)** | `regression or benchmark` | Quick regression tests to catch breaking changes | Weekly (Sunday 2 AM UTC) | -| **[Full Benchmark](./history/full/index.md)** | `easy or medium or hard or regression or benchmark` | Comprehensive testing across all difficulty levels | Manual / On-demand | +| ⚡ **Fast** | `regression or benchmark` | Quick regression tests to catch breaking changes | Weekly (Sunday 2 AM UTC) | +| **Full** | `easy or medium or hard or regression or benchmark` | Comprehensive testing across all difficulty levels | Manual / On-demand | + +All results are stored in [History](./history/index.md). Fast benchmark results are marked with ⚡. ## Test Categories -- **Regression tests (`regression`, `benchmark`)**: Critical scenarios that must always pass +- **Regression tests (`regression`)**: Critical scenarios that must always pass +- **Benchmark tests (`benchmark`)**: Tests included in the fast benchmark for quick validation - **Easy tests (`easy`)**: Straightforward scenarios for baseline validation - **Medium tests (`medium`)**: Moderately complex troubleshooting scenarios - **Hard tests (`hard`)**: Challenging multi-step investigations diff --git a/docs/development/evaluations/latest-results.md b/docs/development/evaluations/latest-results.md index 39f863ff4a..d874fd13a4 100644 --- a/docs/development/evaluations/latest-results.md +++ b/docs/development/evaluations/latest-results.md @@ -1,7 +1,7 @@ -# HolmesGPT LLM Evaluation Fast Benchmark Results +# ⚡ HolmesGPT LLM Evaluation Fast Benchmark Results -**Generated**: 2026-01-04 18:10 UTC -**Total Duration**: 1m 18s +**Generated**: 2026-01-05 15:39 UTC +**Total Duration**: 3m 4s **Iterations**: 1 **Judge (classifier) model**: gpt-4.1 @@ -18,31 +18,42 @@ If you find scenarios that HolmesGPT does not perform well on, please consider a | Model | Pass | Fail | Skip/Error | Total | Success Rate | |-------|------|------|------------|-------|--------------| +| deepseek-3.1 | 1 | 0 | 0 | 1 | 🟢 100% (1/1) | +| gpt-5 | 0 | 1 | 0 | 1 | 🔴 0% (0/1) | +| gpt-5.1 | 1 | 0 | 0 | 1 | 🟢 100% (1/1) | +| haiku-4.5 | 1 | 0 | 0 | 1 | 🟢 100% (1/1) | | sonnet-4.5 | 1 | 0 | 0 | 1 | 🟢 100% (1/1) | ## Model Cost Comparison | Model | Tests | Avg Cost | Min Cost | Max Cost | Total Cost | |-------|-------|----------|----------|----------|------------| -| sonnet-4.5 | 1 | $0.18 | $0.18 | $0.18 | $0.18 | +| gpt-5 | 1 | $0.03 | $0.03 | $0.03 | $0.03 | +| gpt-5.1 | 1 | $0.12 | $0.12 | $0.12 | $0.12 | +| haiku-4.5 | 1 | $0.05 | $0.05 | $0.05 | $0.05 | +| sonnet-4.5 | 1 | $0.19 | $0.19 | $0.19 | $0.19 | ## Model Latency Comparison | Model | Avg (s) | Min (s) | Max (s) | P50 (s) | P95 (s) | |-------|---------|---------|---------|---------|---------| -| sonnet-4.5 | 40.5 | 40.5 | 40.5 | 40.5 | 40.5 | +| deepseek-3.1 | 131.8 | 131.8 | 131.8 | 131.8 | 131.8 | +| gpt-5 | 23.4 | 23.4 | 23.4 | 23.4 | 23.4 | +| gpt-5.1 | 73.1 | 73.1 | 73.1 | 73.1 | 73.1 | +| haiku-4.5 | 31.0 | 31.0 | 31.0 | 31.0 | 31.0 | +| sonnet-4.5 | 47.8 | 47.8 | 47.8 | 47.8 | 47.8 | ## Performance by Tag Success rate by test category and model: -| Tag | sonnet-4.5 | Warnings | -|-----|-------|----------| -| [easy](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522easy%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520easy%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | -| [kubernetes](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522kubernetes%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520kubernetes%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | -| [one-test](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522one-test%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520one-test%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | -| [regression](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522regression%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520regression%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | -| **Overall** | 🟢 100% (1/1) | | +| Tag | deepseek-3.1 | gpt-5 | gpt-5.1 | haiku-4.5 | sonnet-4.5 | Warnings | +|-----|-------|-------|-------|-------|-------|----------| +| [easy](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522easy%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520easy%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | | +| [kubernetes](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522kubernetes%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520kubernetes%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | | +| [one-test](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522one-test%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520one-test%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | | +| [regression](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522regression%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520regression%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | | +| **Overall** | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | | ## Raw Results @@ -56,16 +67,16 @@ Status of all evaluations across models. Color coding: - ⏱️ Timeout or rate limit error - ⏭️ Test skipped (e.g., known issue or precondition not met) -| Eval ID | [sonnet-4.5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522sonnet-4.5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520sonnet-4.5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | -|---------|-------| -| [**09_crashpod**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | -| **SUMMARY** | 🟢 100% (1/1) | +| Eval ID | [deepseek-3.1](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522deepseek-3.1%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520deepseek-3.1%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [gpt-5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522gpt-5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520gpt-5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [gpt-5.1](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522gpt-5.1%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520gpt-5.1%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [haiku-4.5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522haiku-4.5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520haiku-4.5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [sonnet-4.5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522sonnet-4.5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520sonnet-4.5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +|---------|-------|-------|-------|-------|-------| +| [**09_crashpod**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +| **SUMMARY** | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | ## Detailed Raw Results -| Eval ID | sonnet-4.5 | -|---------|-------| -| [09_crashpod](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 40.5s / 💰 $0.18 | +| Eval ID | deepseek-3.1 | gpt-5 | gpt-5.1 | haiku-4.5 | sonnet-4.5 | +|---------|-------|-------|-------|-------|-------| +| [09_crashpod](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 131.8s | [🔴 0% (0/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 23.4s / 💰 $0.03 | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 73.1s / 💰 $0.12 | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 31.0s / 💰 $0.05 | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 47.8s / 💰 $0.19 | --- -*Results are automatically generated and updated weekly. View full traces and detailed analysis in [Braintrust experiment: local-benchmark-20260104-174426](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260104-174426).* +*Results are automatically generated and updated weekly. View full traces and detailed analysis in [Braintrust experiment: local-benchmark-20260105-153457](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457).* diff --git a/run_benchmarks_local.py b/run_benchmarks_local.py index 129a8b6879..5e4476540d 100755 --- a/run_benchmarks_local.py +++ b/run_benchmarks_local.py @@ -209,21 +209,18 @@ def generate_report(self) -> None: # Create output directories based on benchmark type docs_dir = Path("docs/development/evaluations") + history_dir = docs_dir / "history" + history_dir.mkdir(parents=True, exist_ok=True) - # Determine output file and history directory based on benchmark type + # Determine output file based on benchmark type + # History files always use results_TIMESTAMP.md format (⚡ in title distinguishes fast) if self.benchmark_type == "fast-benchmark": main_output = docs_dir / "fast-benchmark-results.md" - history_subdir = "fast" elif self.benchmark_type == "full-benchmark": main_output = docs_dir / "full-benchmark-results.md" - history_subdir = "full" else: # Custom markers - use full benchmark output location main_output = docs_dir / "full-benchmark-results.md" - history_subdir = "full" - - history_dir = docs_dir / "history" / history_subdir - history_dir.mkdir(parents=True, exist_ok=True) # Build base command cmd = [ @@ -247,13 +244,12 @@ def generate_report(self) -> None: subprocess.run(cmd, check=True) print(f"✅ Report generated: {main_output}") - # For fast-benchmark, also copy to latest-results.md for backwards compatibility - if self.benchmark_type == "fast-benchmark": - latest_output = docs_dir / "latest-results.md" - shutil.copy(main_output, latest_output) - print(f"📋 Copied to: {latest_output} (backwards compatibility)") + # Always copy to latest-results.md so it shows whichever benchmark ran most recently + latest_output = docs_dir / "latest-results.md" + shutil.copy(main_output, latest_output) + print(f"📋 Updated: {latest_output}") - # Generate historical copy + # Generate historical copy (⚡ in title distinguishes fast benchmarks) timestamp = datetime.now().strftime("%Y%m%d_%H%M%S") history_output = history_dir / f"results_{timestamp}.md" @@ -302,17 +298,9 @@ def show_summary(self) -> None: ("eval_results.json", "JSON results"), ("evals_report.md", "Evaluation report"), (result_file, "Benchmark results"), + ("docs/development/evaluations/latest-results.md", "Latest results"), ] - # For fast-benchmark, also check latest-results.md - if self.benchmark_type == "fast-benchmark": - files_to_check.append( - ( - "docs/development/evaluations/latest-results.md", - "Latest results (copy)", - ) - ) - for filename, description in files_to_check: path = Path(filename) if path.exists(): @@ -325,8 +313,7 @@ def show_summary(self) -> None: print() print("To commit results (like CI/CD would on main):") print(f" git add {result_file}") - if self.benchmark_type == "fast-benchmark": - print(" git add docs/development/evaluations/latest-results.md") + print(" git add docs/development/evaluations/latest-results.md") print(" git commit -m 'Update benchmark results [skip ci]'") print("=" * 50) diff --git a/tests/generate_eval_report.py b/tests/generate_eval_report.py index ab3ab17762..3cf57b22ef 100755 --- a/tests/generate_eval_report.py +++ b/tests/generate_eval_report.py @@ -1334,28 +1334,34 @@ def main(): benchmark_type = getattr(args, "benchmark_type", None) if benchmark_type == "fast-benchmark": benchmark_label = "Fast Benchmark" + benchmark_icon = "⚡ " elif benchmark_type == "full-benchmark": benchmark_label = "Full Benchmark" + benchmark_icon = "" else: benchmark_label = "Benchmark" + benchmark_icon = "" # Header - check if output file is in history format output_filename = Path(args.output_file).name import re - match = re.match( + # Match results_TIMESTAMP.md pattern for history files + history_match = re.match( r"results_(\d{4})(\d{2})(\d{2})_(\d{2})(\d{2})(\d{2})\.md", output_filename ) - if match: - # Extract date components and format as title - year, month, day, hour, minute, second = match.groups() + + if history_match: + # History file - use date as title, add ⚡ for fast benchmarks + year, month, day, hour, minute, second = history_match.groups() date_obj = datetime(int(year), int(month), int(day)) - # Format as "Month Day, Year" (no time) title = date_obj.strftime("%B %d, %Y") - report_lines.append(f"# {title}") + report_lines.append(f"# {benchmark_icon}{title}") else: - # Default title with benchmark type - report_lines.append(f"# HolmesGPT LLM Evaluation {benchmark_label} Results") + # Default title with benchmark type (for main result files) + report_lines.append( + f"# {benchmark_icon}HolmesGPT LLM Evaluation {benchmark_label} Results" + ) report_lines.append("") # Format duration nicely duration_seconds = results.get("duration", 0) From 95458fa74cedfe61113eff0b2cf8bde516c51182 Mon Sep 17 00:00:00 2001 From: Tomer Keshet Date: Mon, 5 Jan 2026 16:10:57 +0200 Subject: [PATCH 3/7] redirecting to the latest benchmark instead of duplicating the results Signed-off-by: Tomer Keshet --- .../evaluations/fast-benchmark-results.md | 45 ++++------ .../evaluations/full-benchmark-results.md | 71 ++++++++++++++++ .../history/results_20260105_153903.md | 82 ------------------ .../development/evaluations/latest-results.md | 85 ++----------------- run_benchmarks_local.py | 27 ++++-- 5 files changed, 114 insertions(+), 196 deletions(-) create mode 100644 docs/development/evaluations/full-benchmark-results.md delete mode 100644 docs/development/evaluations/history/results_20260105_153903.md diff --git a/docs/development/evaluations/fast-benchmark-results.md b/docs/development/evaluations/fast-benchmark-results.md index d874fd13a4..8c5ba80e7c 100644 --- a/docs/development/evaluations/fast-benchmark-results.md +++ b/docs/development/evaluations/fast-benchmark-results.md @@ -1,7 +1,7 @@ # ⚡ HolmesGPT LLM Evaluation Fast Benchmark Results -**Generated**: 2026-01-05 15:39 UTC -**Total Duration**: 3m 4s +**Generated**: 2026-01-05 16:05 UTC +**Total Duration**: 1m 27s **Iterations**: 1 **Judge (classifier) model**: gpt-4.1 @@ -18,42 +18,31 @@ If you find scenarios that HolmesGPT does not perform well on, please consider a | Model | Pass | Fail | Skip/Error | Total | Success Rate | |-------|------|------|------------|-------|--------------| -| deepseek-3.1 | 1 | 0 | 0 | 1 | 🟢 100% (1/1) | -| gpt-5 | 0 | 1 | 0 | 1 | 🔴 0% (0/1) | -| gpt-5.1 | 1 | 0 | 0 | 1 | 🟢 100% (1/1) | -| haiku-4.5 | 1 | 0 | 0 | 1 | 🟢 100% (1/1) | | sonnet-4.5 | 1 | 0 | 0 | 1 | 🟢 100% (1/1) | ## Model Cost Comparison | Model | Tests | Avg Cost | Min Cost | Max Cost | Total Cost | |-------|-------|----------|----------|----------|------------| -| gpt-5 | 1 | $0.03 | $0.03 | $0.03 | $0.03 | -| gpt-5.1 | 1 | $0.12 | $0.12 | $0.12 | $0.12 | -| haiku-4.5 | 1 | $0.05 | $0.05 | $0.05 | $0.05 | | sonnet-4.5 | 1 | $0.19 | $0.19 | $0.19 | $0.19 | ## Model Latency Comparison | Model | Avg (s) | Min (s) | Max (s) | P50 (s) | P95 (s) | |-------|---------|---------|---------|---------|---------| -| deepseek-3.1 | 131.8 | 131.8 | 131.8 | 131.8 | 131.8 | -| gpt-5 | 23.4 | 23.4 | 23.4 | 23.4 | 23.4 | -| gpt-5.1 | 73.1 | 73.1 | 73.1 | 73.1 | 73.1 | -| haiku-4.5 | 31.0 | 31.0 | 31.0 | 31.0 | 31.0 | | sonnet-4.5 | 47.8 | 47.8 | 47.8 | 47.8 | 47.8 | ## Performance by Tag Success rate by test category and model: -| Tag | deepseek-3.1 | gpt-5 | gpt-5.1 | haiku-4.5 | sonnet-4.5 | Warnings | -|-----|-------|-------|-------|-------|-------|----------| -| [easy](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522easy%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520easy%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | | -| [kubernetes](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522kubernetes%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520kubernetes%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | | -| [one-test](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522one-test%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520one-test%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | | -| [regression](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522regression%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520regression%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | | -| **Overall** | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | | +| Tag | sonnet-4.5 | Warnings | +|-----|-------|----------| +| [easy](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-160330?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522easy%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520easy%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | +| [kubernetes](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-160330?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522kubernetes%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520kubernetes%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | +| [one-test](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-160330?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522one-test%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520one-test%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | +| [regression](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-160330?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522regression%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520regression%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | +| **Overall** | 🟢 100% (1/1) | | ## Raw Results @@ -67,16 +56,16 @@ Status of all evaluations across models. Color coding: - ⏱️ Timeout or rate limit error - ⏭️ Test skipped (e.g., known issue or precondition not met) -| Eval ID | [deepseek-3.1](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522deepseek-3.1%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520deepseek-3.1%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [gpt-5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522gpt-5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520gpt-5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [gpt-5.1](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522gpt-5.1%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520gpt-5.1%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [haiku-4.5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522haiku-4.5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520haiku-4.5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [sonnet-4.5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522sonnet-4.5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520sonnet-4.5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | -|---------|-------|-------|-------|-------|-------| -| [**09_crashpod**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | -| **SUMMARY** | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | +| Eval ID | [sonnet-4.5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-160330?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522sonnet-4.5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520sonnet-4.5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +|---------|-------| +| [**09_crashpod**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-160330?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-160330?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +| **SUMMARY** | 🟢 100% (1/1) | ## Detailed Raw Results -| Eval ID | deepseek-3.1 | gpt-5 | gpt-5.1 | haiku-4.5 | sonnet-4.5 | -|---------|-------|-------|-------|-------|-------| -| [09_crashpod](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 131.8s | [🔴 0% (0/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 23.4s / 💰 $0.03 | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 73.1s / 💰 $0.12 | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 31.0s / 💰 $0.05 | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 47.8s / 💰 $0.19 | +| Eval ID | sonnet-4.5 | +|---------|-------| +| [09_crashpod](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-160330?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-160330?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 47.8s / 💰 $0.19 | --- -*Results are automatically generated and updated weekly. View full traces and detailed analysis in [Braintrust experiment: local-benchmark-20260105-153457](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457).* +*Results are automatically generated and updated weekly. View full traces and detailed analysis in [Braintrust experiment: local-benchmark-20260105-160330](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-160330).* diff --git a/docs/development/evaluations/full-benchmark-results.md b/docs/development/evaluations/full-benchmark-results.md new file mode 100644 index 0000000000..ec58877ef9 --- /dev/null +++ b/docs/development/evaluations/full-benchmark-results.md @@ -0,0 +1,71 @@ +# HolmesGPT LLM Evaluation Full Benchmark Results + +**Generated**: 2026-01-05 16:08 UTC +**Total Duration**: 1m 27s +**Iterations**: 1 +**Judge (classifier) model**: gpt-4.1 + +!!! info "Full Benchmark" + **Markers**: `easy or medium or hard or regression or benchmark`
+ **Schedule**: Manual / On-demand
+ **Purpose**: Comprehensive testing across all difficulty levels + +HolmesGPT is continuously evaluated against real-world Kubernetes and cloud troubleshooting scenarios. + +If you find scenarios that HolmesGPT does not perform well on, please consider adding them as evals to the benchmark. + +## Model Accuracy Comparison + +| Model | Pass | Fail | Skip/Error | Total | Success Rate | +|-------|------|------|------------|-------|--------------| +| sonnet-4.5 | 1 | 0 | 0 | 1 | 🟢 100% (1/1) | + +## Model Cost Comparison + +| Model | Tests | Avg Cost | Min Cost | Max Cost | Total Cost | +|-------|-------|----------|----------|----------|------------| +| sonnet-4.5 | 1 | $0.15 | $0.15 | $0.15 | $0.15 | + +## Model Latency Comparison + +| Model | Avg (s) | Min (s) | Max (s) | P50 (s) | P95 (s) | +|-------|---------|---------|---------|---------|---------| +| sonnet-4.5 | 49.8 | 49.8 | 49.8 | 49.8 | 49.8 | + +## Performance by Tag + +Success rate by test category and model: + +| Tag | sonnet-4.5 | Warnings | +|-----|-------|----------| +| [easy](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-160651?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522easy%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520easy%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | +| [kubernetes](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-160651?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522kubernetes%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520kubernetes%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | +| [one-test](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-160651?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522one-test%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520one-test%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | +| [regression](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-160651?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522regression%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520regression%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | | +| **Overall** | 🟢 100% (1/1) | | + +## Raw Results + +Status of all evaluations across models. Color coding: + +- 🟢 Passing 100% (stable) +- 🟡 Passing 1-99% +- 🔴 Passing 0% (failing) +- 🔧 Mock data failure (missing or invalid test data) +- ⚠️ Setup failure (environment/infrastructure issue) +- ⏱️ Timeout or rate limit error +- ⏭️ Test skipped (e.g., known issue or precondition not met) + +| Eval ID | [sonnet-4.5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-160651?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522sonnet-4.5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520sonnet-4.5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +|---------|-------| +| [**09_crashpod**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-160651?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-160651?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | +| **SUMMARY** | 🟢 100% (1/1) | + +## Detailed Raw Results + +| Eval ID | sonnet-4.5 | +|---------|-------| +| [09_crashpod](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-160651?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-160651?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 49.8s / 💰 $0.15 | + +--- +*Results are automatically generated and updated weekly. View full traces and detailed analysis in [Braintrust experiment: local-benchmark-20260105-160651](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-160651).* diff --git a/docs/development/evaluations/history/results_20260105_153903.md b/docs/development/evaluations/history/results_20260105_153903.md deleted file mode 100644 index b7ea324c64..0000000000 --- a/docs/development/evaluations/history/results_20260105_153903.md +++ /dev/null @@ -1,82 +0,0 @@ -# ⚡ January 05, 2026 - -**Generated**: 2026-01-05 15:39 UTC -**Total Duration**: 3m 4s -**Iterations**: 1 -**Judge (classifier) model**: gpt-4.1 - -!!! info "Fast Benchmark" - **Markers**: `regression or benchmark`
- **Schedule**: Weekly (Sunday 2 AM UTC)
- **Purpose**: Quick regression tests to catch breaking changes - -HolmesGPT is continuously evaluated against real-world Kubernetes and cloud troubleshooting scenarios. - -If you find scenarios that HolmesGPT does not perform well on, please consider adding them as evals to the benchmark. - -## Model Accuracy Comparison - -| Model | Pass | Fail | Skip/Error | Total | Success Rate | -|-------|------|------|------------|-------|--------------| -| deepseek-3.1 | 1 | 0 | 0 | 1 | 🟢 100% (1/1) | -| gpt-5 | 0 | 1 | 0 | 1 | 🔴 0% (0/1) | -| gpt-5.1 | 1 | 0 | 0 | 1 | 🟢 100% (1/1) | -| haiku-4.5 | 1 | 0 | 0 | 1 | 🟢 100% (1/1) | -| sonnet-4.5 | 1 | 0 | 0 | 1 | 🟢 100% (1/1) | - -## Model Cost Comparison - -| Model | Tests | Avg Cost | Min Cost | Max Cost | Total Cost | -|-------|-------|----------|----------|----------|------------| -| gpt-5 | 1 | $0.03 | $0.03 | $0.03 | $0.03 | -| gpt-5.1 | 1 | $0.12 | $0.12 | $0.12 | $0.12 | -| haiku-4.5 | 1 | $0.05 | $0.05 | $0.05 | $0.05 | -| sonnet-4.5 | 1 | $0.19 | $0.19 | $0.19 | $0.19 | - -## Model Latency Comparison - -| Model | Avg (s) | Min (s) | Max (s) | P50 (s) | P95 (s) | -|-------|---------|---------|---------|---------|---------| -| deepseek-3.1 | 131.8 | 131.8 | 131.8 | 131.8 | 131.8 | -| gpt-5 | 23.4 | 23.4 | 23.4 | 23.4 | 23.4 | -| gpt-5.1 | 73.1 | 73.1 | 73.1 | 73.1 | 73.1 | -| haiku-4.5 | 31.0 | 31.0 | 31.0 | 31.0 | 31.0 | -| sonnet-4.5 | 47.8 | 47.8 | 47.8 | 47.8 | 47.8 | - -## Performance by Tag - -Success rate by test category and model: - -| Tag | deepseek-3.1 | gpt-5 | gpt-5.1 | haiku-4.5 | sonnet-4.5 | Warnings | -|-----|-------|-------|-------|-------|-------|----------| -| [easy](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522easy%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520easy%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | | -| [kubernetes](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522kubernetes%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520kubernetes%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | | -| [one-test](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522one-test%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520one-test%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | | -| [regression](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522regression%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520regression%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | | -| **Overall** | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | | - -## Raw Results - -Status of all evaluations across models. Color coding: - -- 🟢 Passing 100% (stable) -- 🟡 Passing 1-99% -- 🔴 Passing 0% (failing) -- 🔧 Mock data failure (missing or invalid test data) -- ⚠️ Setup failure (environment/infrastructure issue) -- ⏱️ Timeout or rate limit error -- ⏭️ Test skipped (e.g., known issue or precondition not met) - -| Eval ID | [deepseek-3.1](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522deepseek-3.1%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520deepseek-3.1%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [gpt-5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522gpt-5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520gpt-5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [gpt-5.1](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522gpt-5.1%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520gpt-5.1%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [haiku-4.5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522haiku-4.5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520haiku-4.5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [sonnet-4.5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522sonnet-4.5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520sonnet-4.5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | -|---------|-------|-------|-------|-------|-------| -| [**09_crashpod**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | -| **SUMMARY** | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | - -## Detailed Raw Results - -| Eval ID | deepseek-3.1 | gpt-5 | gpt-5.1 | haiku-4.5 | sonnet-4.5 | -|---------|-------|-------|-------|-------|-------| -| [09_crashpod](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 131.8s | [🔴 0% (0/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 23.4s / 💰 $0.03 | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 73.1s / 💰 $0.12 | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 31.0s / 💰 $0.05 | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 47.8s / 💰 $0.19 | - ---- -*Results are automatically generated and updated weekly. View full traces and detailed analysis in [Braintrust experiment: local-benchmark-20260105-153457](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457).* diff --git a/docs/development/evaluations/latest-results.md b/docs/development/evaluations/latest-results.md index d874fd13a4..8c8d7bdc57 100644 --- a/docs/development/evaluations/latest-results.md +++ b/docs/development/evaluations/latest-results.md @@ -1,82 +1,9 @@ -# ⚡ HolmesGPT LLM Evaluation Fast Benchmark Results +# Latest Results -**Generated**: 2026-01-05 15:39 UTC -**Total Duration**: 3m 4s -**Iterations**: 1 -**Judge (classifier) model**: gpt-4.1 +Redirecting to the latest benchmark results... -!!! info "Fast Benchmark" - **Markers**: `regression or benchmark`
- **Schedule**: Weekly (Sunday 2 AM UTC)
- **Purpose**: Quick regression tests to catch breaking changes + -HolmesGPT is continuously evaluated against real-world Kubernetes and cloud troubleshooting scenarios. - -If you find scenarios that HolmesGPT does not perform well on, please consider adding them as evals to the benchmark. - -## Model Accuracy Comparison - -| Model | Pass | Fail | Skip/Error | Total | Success Rate | -|-------|------|------|------------|-------|--------------| -| deepseek-3.1 | 1 | 0 | 0 | 1 | 🟢 100% (1/1) | -| gpt-5 | 0 | 1 | 0 | 1 | 🔴 0% (0/1) | -| gpt-5.1 | 1 | 0 | 0 | 1 | 🟢 100% (1/1) | -| haiku-4.5 | 1 | 0 | 0 | 1 | 🟢 100% (1/1) | -| sonnet-4.5 | 1 | 0 | 0 | 1 | 🟢 100% (1/1) | - -## Model Cost Comparison - -| Model | Tests | Avg Cost | Min Cost | Max Cost | Total Cost | -|-------|-------|----------|----------|----------|------------| -| gpt-5 | 1 | $0.03 | $0.03 | $0.03 | $0.03 | -| gpt-5.1 | 1 | $0.12 | $0.12 | $0.12 | $0.12 | -| haiku-4.5 | 1 | $0.05 | $0.05 | $0.05 | $0.05 | -| sonnet-4.5 | 1 | $0.19 | $0.19 | $0.19 | $0.19 | - -## Model Latency Comparison - -| Model | Avg (s) | Min (s) | Max (s) | P50 (s) | P95 (s) | -|-------|---------|---------|---------|---------|---------| -| deepseek-3.1 | 131.8 | 131.8 | 131.8 | 131.8 | 131.8 | -| gpt-5 | 23.4 | 23.4 | 23.4 | 23.4 | 23.4 | -| gpt-5.1 | 73.1 | 73.1 | 73.1 | 73.1 | 73.1 | -| haiku-4.5 | 31.0 | 31.0 | 31.0 | 31.0 | 31.0 | -| sonnet-4.5 | 47.8 | 47.8 | 47.8 | 47.8 | 47.8 | - -## Performance by Tag - -Success rate by test category and model: - -| Tag | deepseek-3.1 | gpt-5 | gpt-5.1 | haiku-4.5 | sonnet-4.5 | Warnings | -|-----|-------|-------|-------|-------|-------|----------| -| [easy](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522easy%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520easy%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | | -| [kubernetes](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522kubernetes%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520kubernetes%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | | -| [one-test](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522one-test%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520one-test%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | | -| [regression](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22tags%2520includes%2520%255B%2522regression%2522%255D%22%2C%20%22label%22%3A%20%22Tags%2520includes%2520regression%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | | -| **Overall** | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | | - -## Raw Results - -Status of all evaluations across models. Color coding: - -- 🟢 Passing 100% (stable) -- 🟡 Passing 1-99% -- 🔴 Passing 0% (failing) -- 🔧 Mock data failure (missing or invalid test data) -- ⚠️ Setup failure (environment/infrastructure issue) -- ⏱️ Timeout or rate limit error -- ⏭️ Test skipped (e.g., known issue or precondition not met) - -| Eval ID | [deepseek-3.1](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522deepseek-3.1%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520deepseek-3.1%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [gpt-5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522gpt-5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520gpt-5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [gpt-5.1](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522gpt-5.1%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520gpt-5.1%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [haiku-4.5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522haiku-4.5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520haiku-4.5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [sonnet-4.5](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.model%2520%253D%2520%2522sonnet-4.5%2522%22%2C%20%22label%22%3A%20%22metadata.model%2520equals%2520sonnet-4.5%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | -|---------|-------|-------|-------|-------|-------| -| [**09_crashpod**](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🔴](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | -| **SUMMARY** | 🟢 100% (1/1) | 🔴 0% (0/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | 🟢 100% (1/1) | - -## Detailed Raw Results - -| Eval ID | deepseek-3.1 | gpt-5 | gpt-5.1 | haiku-4.5 | sonnet-4.5 | -|---------|-------|-------|-------|-------|-------| -| [09_crashpod](https://github.com/HolmesGPT/holmesgpt/blob/master/tests/llm/fixtures/test_ask_holmes/09_crashpod/test_case.yaml) [🔗](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22metadata.eval_id%2520%253D%2520%252209_crashpod%2522%22%2C%20%22label%22%3A%20%22metadata.eval_id%2520equals%252009_crashpod%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bdeepseek-3.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bdeepseek-3.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 131.8s | [🔴 0% (0/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 23.4s / 💰 $0.03 | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bgpt-5.1%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bgpt-5.1%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 73.1s / 💰 $0.12 | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bhaiku-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bhaiku-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 31.0s / 💰 $0.05 | [🟢 100% (1/1)](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457?c=&search=%7B%22filter%22%3A%20%5B%7B%22text%22%3A%20%22span_attributes.name%2520%253D%2520%252209_crashpod%255Bsonnet-4.5%255D%2522%22%2C%20%22label%22%3A%20%22Name%2520equals%252009_crashpod%255Bsonnet-4.5%255D%22%2C%20%22originType%22%3A%20%22form%22%7D%5D%7D) / ⏱️ 47.8s / 💰 $0.19 | - ---- -*Results are automatically generated and updated weekly. View full traces and detailed analysis in [Braintrust experiment: local-benchmark-20260105-153457](https://www.braintrust.dev/app/robustadev/p/HolmesGPT/experiments/local-benchmark-20260105-153457).* +If you are not redirected automatically, [click here](../history/results_20260104_174301/). diff --git a/run_benchmarks_local.py b/run_benchmarks_local.py index 5e4476540d..c5501338ec 100755 --- a/run_benchmarks_local.py +++ b/run_benchmarks_local.py @@ -8,7 +8,6 @@ import argparse import os -import shutil import subprocess import sys from datetime import datetime @@ -244,12 +243,7 @@ def generate_report(self) -> None: subprocess.run(cmd, check=True) print(f"✅ Report generated: {main_output}") - # Always copy to latest-results.md so it shows whichever benchmark ran most recently - latest_output = docs_dir / "latest-results.md" - shutil.copy(main_output, latest_output) - print(f"📋 Updated: {latest_output}") - - # Generate historical copy (⚡ in title distinguishes fast benchmarks) + # Generate historical copy first (⚡ in title distinguishes fast benchmarks) timestamp = datetime.now().strftime("%Y%m%d_%H%M%S") history_output = history_dir / f"results_{timestamp}.md" @@ -260,6 +254,25 @@ def generate_report(self) -> None: subprocess.run(cmd_history, check=True) print(f"📁 Saved historical copy: {history_output}") + # Create redirect page for latest-results.md pointing to the history file + latest_output = docs_dir / "latest-results.md" + history_relative = ( + f"../history/{history_output.name.replace('.md', '/')}".rstrip("/") + + "/" + ) + redirect_content = f"""# Latest Results + +Redirecting to the latest benchmark results... + + + +If you are not redirected automatically, [click here]({history_relative}). +""" + latest_output.write_text(redirect_content) + print(f"📋 Updated redirect: {latest_output} -> {history_relative}") + except subprocess.CalledProcessError as e: print(f"❌ Report generation failed: {e}") From 356d4a4325f115039488306fec3d4089e3b2ea2d Mon Sep 17 00:00:00 2001 From: Tomer Keshet Date: Mon, 5 Jan 2026 16:12:29 +0200 Subject: [PATCH 4/7] weekly benchmarks github action Signed-off-by: Tomer Keshet --- .github/workflows/eval-benchmarks.yaml | 147 +++++-------------------- 1 file changed, 26 insertions(+), 121 deletions(-) diff --git a/.github/workflows/eval-benchmarks.yaml b/.github/workflows/eval-benchmarks.yaml index 6e7efd5042..fd48f0052d 100644 --- a/.github/workflows/eval-benchmarks.yaml +++ b/.github/workflows/eval-benchmarks.yaml @@ -13,7 +13,7 @@ on: - 'full-benchmark' default: 'fast-benchmark' models: - description: 'Comma-separated list of models to test (leave empty for defaults from run_benchmarks_local.py)' + description: 'Comma-separated list of models to test (leave empty for script defaults)' required: false default: '' test_markers: @@ -54,69 +54,42 @@ jobs: cluster-name: 'kind' wait-for-ready: 'true' - - name: Determine test command - id: test-command + - name: Build benchmark command + id: build-command run: | - # Define benchmark type markers - FAST_MARKERS="regression or benchmark" - FULL_MARKERS="easy or medium or hard or regression or benchmark" - - # Default models from run_benchmarks_local.py - DEFAULT_MODELS="gpt-5.1,gpt-5,sonnet-4.5,haiku-4.5,deepseek-3.1" + # Start with base command + CMD="./run_benchmarks_local.py" if [ "${{ github.event_name }}" == "workflow_dispatch" ]; then BENCHMARK_TYPE="${{ github.event.inputs.benchmark_type }}" CUSTOM_MARKERS="${{ github.event.inputs.test_markers }}" - MODEL="${{ github.event.inputs.models }}" + MODELS="${{ github.event.inputs.models }}" ITERATIONS="${{ github.event.inputs.iterations }}" - # Validate: cannot combine benchmark_type with custom markers - if [ -n "$CUSTOM_MARKERS" ] && [ -n "$BENCHMARK_TYPE" ]; then - echo "ERROR: Cannot combine benchmark_type with custom test_markers" - exit 1 - fi - - # Determine markers based on benchmark type or custom markers + # Add benchmark type or custom markers (script validates they're mutually exclusive) if [ -n "$CUSTOM_MARKERS" ]; then - TEST_MARKERS="llm and ($CUSTOM_MARKERS)" - BENCHMARK_TYPE="" - elif [ "$BENCHMARK_TYPE" == "full-benchmark" ]; then - TEST_MARKERS="llm and ($FULL_MARKERS)" - else - # Default to fast-benchmark - BENCHMARK_TYPE="fast-benchmark" - TEST_MARKERS="llm and ($FAST_MARKERS)" + CMD="$CMD --markers \"$CUSTOM_MARKERS\"" + elif [ -n "$BENCHMARK_TYPE" ]; then + CMD="$CMD --benchmark-type $BENCHMARK_TYPE" fi - # Use default models if not specified - if [ -z "$MODEL" ]; then - MODEL="$DEFAULT_MODELS" + # Add models only if specified (otherwise use script defaults) + if [ -n "$MODELS" ]; then + CMD="$CMD --models $MODELS" fi - # Cap iterations at 10 - if [ "$ITERATIONS" -gt 10 ]; then - echo "Capping iterations at 10 (requested: $ITERATIONS)" - ITERATIONS="10" + # Add iterations if specified + if [ -n "$ITERATIONS" ] && [ "$ITERATIONS" != "1" ]; then + CMD="$CMD --iterations $ITERATIONS" fi - elif [ "${{ github.event_name }}" == "schedule" ]; then - # Weekly scheduled run: fast-benchmark with defaults + else + # Scheduled run: use fast-benchmark with all defaults BENCHMARK_TYPE="fast-benchmark" - MODEL="$DEFAULT_MODELS" - TEST_MARKERS="llm and ($FAST_MARKERS)" - ITERATIONS="1" + CMD="$CMD --benchmark-type fast-benchmark" fi - # Set test path - TEST_PATH="tests/llm/" - - # Write all outputs atomically - { - echo "models=$MODEL" - echo "test_path=$TEST_PATH" - echo "test_markers=$TEST_MARKERS" - echo "iterations=$ITERATIONS" - echo "benchmark_type=$BENCHMARK_TYPE" - } >> $GITHUB_OUTPUT + echo "command=$CMD" >> $GITHUB_OUTPUT + echo "benchmark_type=$BENCHMARK_TYPE" >> $GITHUB_OUTPUT - name: Run evaluation benchmarks env: @@ -125,78 +98,11 @@ jobs: AZURE_API_BASE: ${{ secrets.AZURE_API_BASE }} AZURE_API_KEY: ${{ secrets.AZURE_API_KEY }} AZURE_API_VERSION: ${{ secrets.AZURE_API_VERSION }} - MODEL: ${{ steps.test-command.outputs.models }} - ITERATIONS: ${{ steps.test-command.outputs.iterations }} - RUN_LIVE: "true" - CLASSIFIER_MODEL: "gpt-4o" # Always use OpenAI for classification BRAINTRUST_API_KEY: ${{ secrets.BRAINTRUST_API_KEY }} EXPERIMENT_ID: "ci-benchmark-${{ github.run_id }}" - UPLOAD_DATASET: "true" run: | - poetry run pytest "${{ steps.test-command.outputs.test_path }}" \ - -m "${{ steps.test-command.outputs.test_markers }}" \ - --no-cov \ - --tb=short \ - -v \ - --strict-setup-mode \ - --strict-setup-exceptions=22_high_latency_dbi_down \ - --json-report \ - --json-report-file=eval_results.json \ - || true # Don't fail the workflow if tests fail - - # TODO: For testing, show what would be run - echo "==== Test execution summary ====" - echo "Models: ${{ steps.test-command.outputs.models }}" - echo "Markers: ${{ steps.test-command.outputs.test_markers }}" - echo "Iterations: ${{ steps.test-command.outputs.iterations }}" - echo "================================" - - - name: Generate benchmark report - if: always() - run: | - BENCHMARK_TYPE="${{ steps.test-command.outputs.benchmark_type }}" - TIMESTAMP=$(date +%Y%m%d_%H%M%S) - - # Determine output paths based on benchmark type - if [ "$BENCHMARK_TYPE" == "fast-benchmark" ]; then - MAIN_OUTPUT="docs/development/evaluations/fast-benchmark-results.md" - HISTORY_DIR="docs/development/evaluations/history/fast" - elif [ "$BENCHMARK_TYPE" == "full-benchmark" ]; then - MAIN_OUTPUT="docs/development/evaluations/full-benchmark-results.md" - HISTORY_DIR="docs/development/evaluations/history/full" - else - # Custom markers - use full benchmark location - MAIN_OUTPUT="docs/development/evaluations/full-benchmark-results.md" - HISTORY_DIR="docs/development/evaluations/history/full" - fi - - # Create history directory if needed - mkdir -p "$HISTORY_DIR" - - # Build benchmark-type argument if specified - BENCHMARK_ARG="" - if [ -n "$BENCHMARK_TYPE" ]; then - BENCHMARK_ARG="--benchmark-type $BENCHMARK_TYPE" - fi - - # Generate main results - poetry run python tests/generate_eval_report.py \ - --json-file eval_results.json \ - --output-file "$MAIN_OUTPUT" \ - --models "${{ steps.test-command.outputs.models }}" \ - $BENCHMARK_ARG - - # For fast-benchmark, also copy to latest-results.md for backwards compatibility - if [ "$BENCHMARK_TYPE" == "fast-benchmark" ]; then - cp "$MAIN_OUTPUT" docs/development/evaluations/latest-results.md - fi - - # Generate timestamped version for history - poetry run python tests/generate_eval_report.py \ - --json-file eval_results.json \ - --output-file "${HISTORY_DIR}/results_${TIMESTAMP}.md" \ - --models "${{ steps.test-command.outputs.models }}" \ - $BENCHMARK_ARG + echo "Running: ${{ steps.build-command.outputs.command }}" + ${{ steps.build-command.outputs.command }} - name: Upload eval results if: always() @@ -207,6 +113,7 @@ jobs: docs/development/evaluations/fast-benchmark-results.md docs/development/evaluations/full-benchmark-results.md docs/development/evaluations/latest-results.md + eval_results.json - name: Create PR with benchmark results if: always() && (github.event_name == 'schedule' || github.event_name == 'workflow_dispatch') @@ -231,7 +138,7 @@ jobs: fi # Commit and push - BENCHMARK_TYPE="${{ steps.test-command.outputs.benchmark_type }}" + BENCHMARK_TYPE="${{ steps.build-command.outputs.benchmark_type }}" git commit -m "Update ${BENCHMARK_TYPE:-benchmark} results [skip ci]" git push origin "$BRANCH_NAME" --force @@ -240,9 +147,7 @@ jobs: PR_BODY="Automated weekly benchmark results from CI. **Benchmark Type**: ${BENCHMARK_TYPE:-custom} - **Models**: ${{ steps.test-command.outputs.models }} - **Iterations**: ${{ steps.test-command.outputs.iterations }} - **Markers**: ${{ steps.test-command.outputs.test_markers }}" + **Command**: ${{ steps.build-command.outputs.command }}" # Check if PR already exists EXISTING_PR=$(gh pr list --head "$BRANCH_NAME" --json number --jq '.[0].number') From 8bb10abce91bb978d706f94d3da281cd7188164a Mon Sep 17 00:00:00 2001 From: Tomer Keshet Date: Mon, 5 Jan 2026 17:36:03 +0200 Subject: [PATCH 5/7] making the eval-bechamark technically run but only with gpt-4.1 Signed-off-by: Tomer Keshet --- .github/workflows/eval-benchmarks.yaml | 8 +- conftest.py | 4 +- holmes/plugins/toolsets/__init__.py | 14 +- .../toolsets/elasticsearch/elasticsearch.py | 164 +++++++++++------- holmes/plugins/toolsets/grafana/common.py | 4 - tests/generate_eval_report.py | 2 +- 6 files changed, 115 insertions(+), 81 deletions(-) diff --git a/.github/workflows/eval-benchmarks.yaml b/.github/workflows/eval-benchmarks.yaml index fd48f0052d..04be5f41db 100644 --- a/.github/workflows/eval-benchmarks.yaml +++ b/.github/workflows/eval-benchmarks.yaml @@ -13,9 +13,9 @@ on: - 'full-benchmark' default: 'fast-benchmark' models: - description: 'Comma-separated list of models to test (leave empty for script defaults)' + description: 'Comma-separated list of models to test' required: false - default: '' + default: 'gpt-4.1' test_markers: description: 'Custom pytest markers (ONLY use if not using benchmark_type). Cannot be combined with benchmark_type.' required: false @@ -83,9 +83,9 @@ jobs: CMD="$CMD --iterations $ITERATIONS" fi else - # Scheduled run: use fast-benchmark with all defaults + # Scheduled run: use fast-benchmark with gpt-4.1 BENCHMARK_TYPE="fast-benchmark" - CMD="$CMD --benchmark-type fast-benchmark" + CMD="$CMD --benchmark-type fast-benchmark --models gpt-4.1" fi echo "command=$CMD" >> $GITHUB_OUTPUT diff --git a/conftest.py b/conftest.py index a48c0768c5..802bfd8a56 100644 --- a/conftest.py +++ b/conftest.py @@ -233,7 +233,9 @@ def responses(): # Allow Elasticsearch/OpenSearch Cloud API calls (various hosting regions) rsps.add_passthru(re.compile(r"https://.*\.cloud\.es\.io")) # Elastic Cloud rsps.add_passthru(re.compile(r"https://.*\.elastic-cloud\.com")) # Azure-hosted - rsps.add_passthru(re.compile(r"https://.*\.es\.amazonaws\.com")) # AWS OpenSearch + rsps.add_passthru( + re.compile(r"https://.*\.es\.amazonaws\.com") + ) # AWS OpenSearch # Allow rsps.add_passthru("https://google.com") diff --git a/holmes/plugins/toolsets/__init__.py b/holmes/plugins/toolsets/__init__.py index acbce3613b..d9ae1cec39 100644 --- a/holmes/plugins/toolsets/__init__.py +++ b/holmes/plugins/toolsets/__init__.py @@ -28,6 +28,13 @@ from holmes.plugins.toolsets.datadog.toolset_datadog_traces import ( DatadogTracesToolset, ) +from holmes.plugins.toolsets.elasticsearch.elasticsearch import ( + ElasticsearchClusterToolset, + ElasticsearchDataToolset, +) +from holmes.plugins.toolsets.elasticsearch.opensearch_query_assist import ( + OpenSearchQueryAssistToolset, +) from holmes.plugins.toolsets.git import GitToolset from holmes.plugins.toolsets.grafana.loki.toolset_grafana_loki import GrafanaLokiToolset from holmes.plugins.toolsets.grafana.toolset_grafana import GrafanaToolset @@ -41,19 +48,12 @@ from holmes.plugins.toolsets.kubernetes_logs import KubernetesLogsToolset from holmes.plugins.toolsets.mcp.toolset_mcp import RemoteMCPToolset from holmes.plugins.toolsets.newrelic.newrelic import NewRelicToolset -from holmes.plugins.toolsets.elasticsearch.opensearch_query_assist import ( - OpenSearchQueryAssistToolset, -) from holmes.plugins.toolsets.rabbitmq.toolset_rabbitmq import RabbitMQToolset from holmes.plugins.toolsets.robusta.robusta import RobustaToolset from holmes.plugins.toolsets.runbook.runbook_fetcher import RunbookToolset from holmes.plugins.toolsets.servicenow_tables.servicenow_tables import ( ServiceNowTablesToolset, ) -from holmes.plugins.toolsets.elasticsearch.elasticsearch import ( - ElasticsearchClusterToolset, - ElasticsearchDataToolset, -) THIS_DIR = os.path.abspath(os.path.dirname(__file__)) diff --git a/holmes/plugins/toolsets/elasticsearch/elasticsearch.py b/holmes/plugins/toolsets/elasticsearch/elasticsearch.py index 6751cea81b..234c6cd07b 100644 --- a/holmes/plugins/toolsets/elasticsearch/elasticsearch.py +++ b/holmes/plugins/toolsets/elasticsearch/elasticsearch.py @@ -2,7 +2,7 @@ from abc import ABC from typing import Any, ClassVar, Dict, Optional, Tuple, Type -import requests +import requests # type: ignore[import-untyped] from pydantic import BaseModel, ConfigDict from holmes.core.tools import ( @@ -77,16 +77,31 @@ def _perform_health_check(self) -> Tuple[bool, str]: response = self._make_request("GET", "_cluster/health", timeout=10) cluster_name = response.get("cluster_name", "unknown") status = response.get("status", "unknown") - return True, f"Connected to Elasticsearch cluster '{cluster_name}' (status: {status})" + return ( + True, + f"Connected to Elasticsearch cluster '{cluster_name}' (status: {status})", + ) except requests.exceptions.HTTPError as e: if e.response.status_code == 401: - return False, "Elasticsearch authentication failed. Check your API key or credentials." + return ( + False, + "Elasticsearch authentication failed. Check your API key or credentials.", + ) elif e.response.status_code == 403: - return False, "Elasticsearch access denied. Ensure your credentials have cluster access." + return ( + False, + "Elasticsearch access denied. Ensure your credentials have cluster access.", + ) else: - return False, f"Elasticsearch API error: {e.response.status_code} - {e.response.text}" + return ( + False, + f"Elasticsearch API error: {e.response.status_code} - {e.response.text}", + ) except requests.exceptions.ConnectionError: - return False, f"Failed to connect to Elasticsearch at {self.elasticsearch_config.url}" + return ( + False, + f"Failed to connect to Elasticsearch at {self.elasticsearch_config.url}", + ) except requests.exceptions.Timeout: return False, "Elasticsearch health check timed out" except Exception as e: @@ -118,7 +133,10 @@ def _get_headers(self) -> Dict[str, str]: def _get_auth(self) -> Optional[Tuple[str, str]]: """Return basic auth tuple if username/password configured.""" if self.elasticsearch_config.username and self.elasticsearch_config.password: - return (self.elasticsearch_config.username, self.elasticsearch_config.password) + return ( + self.elasticsearch_config.username, + self.elasticsearch_config.password, + ) return None def _make_request( @@ -290,7 +308,13 @@ def _invoke(self, params: dict, context: ToolInvokeContext) -> StructuredToolRes index = params.get("index") # Build the endpoint path - if index and endpoint in ("shards", "indices", "segments", "recovery", "aliases"): + if index and endpoint in ( + "shards", + "indices", + "segments", + "recovery", + "aliases", + ): path = f"_cat/{endpoint}/{index}" else: path = f"_cat/{endpoint}" @@ -313,7 +337,9 @@ def get_parameterized_one_liner(self, params: Dict) -> str: endpoint = params.get("endpoint", "") index = params.get("index", "") suffix = f" ({index})" if index else "" - return f"{toolset_name_for_one_liner(self._toolset.name)}: Cat {endpoint}{suffix}" + return ( + f"{toolset_name_for_one_liner(self._toolset.name)}: Cat {endpoint}{suffix}" + ) class ElasticsearchSearch(BaseElasticsearchTool): @@ -468,7 +494,9 @@ def _invoke(self, params: dict, context: ToolInvokeContext) -> StructuredToolRes def get_parameterized_one_liner(self, params: Dict) -> str: index = params.get("index", "") suffix = f" ({index})" if index else "" - return f"{toolset_name_for_one_liner(self._toolset.name)}: Cluster health{suffix}" + return ( + f"{toolset_name_for_one_liner(self._toolset.name)}: Cluster health{suffix}" + ) class ElasticsearchMappings(BaseElasticsearchTool, JsonFilterMixin): @@ -485,13 +513,15 @@ def __init__(self, toolset: ElasticsearchBaseToolset): "For large mappings, use the jq parameter to filter results " "(e.g., jq='.*.mappings.properties | keys' to list field names)." ), - parameters=JsonFilterMixin.extend_parameters({ - "index": ToolParameter( - description="Index name or pattern to get mappings for", - type="string", - required=True, - ), - }), + parameters=JsonFilterMixin.extend_parameters( + { + "index": ToolParameter( + description="Index name or pattern to get mappings for", + type="string", + required=True, + ), + } + ), ) def _invoke(self, params: dict, context: ToolInvokeContext) -> StructuredToolResult: @@ -591,7 +621,9 @@ def _invoke(self, params: dict, context: ToolInvokeContext) -> StructuredToolRes "primary": params.get("primary", True), } - return self._make_request("GET", "_cluster/allocation/explain", params, body=body) + return self._make_request( + "GET", "_cluster/allocation/explain", params, body=body + ) def get_parameterized_one_liner(self, params: Dict) -> str: index = params.get("index", "") @@ -642,7 +674,9 @@ def _invoke(self, params: dict, context: ToolInvokeContext) -> StructuredToolRes def get_parameterized_one_liner(self, params: Dict) -> str: node_id = params.get("node_id", "_all") - return f"{toolset_name_for_one_liner(self._toolset.name)}: Node stats ({node_id})" + return ( + f"{toolset_name_for_one_liner(self._toolset.name)}: Node stats ({node_id})" + ) class ElasticsearchListIndices(BaseElasticsearchTool, JsonFilterMixin): @@ -657,54 +691,56 @@ def __init__(self, toolset: ElasticsearchBaseToolset): "Returns index names, document counts, and storage size. " "Supports server-side sorting and filtering for efficient queries on large clusters." ), - parameters=JsonFilterMixin.extend_parameters({ - "pattern": ToolParameter( - description=( - "Index name pattern to match. Supports wildcards (e.g., 'logs-*', 'app-*'). " - "Use '*' to list all indices." + parameters=JsonFilterMixin.extend_parameters( + { + "pattern": ToolParameter( + description=( + "Index name pattern to match. Supports wildcards (e.g., 'logs-*', 'app-*'). " + "Use '*' to list all indices." + ), + type="string", + required=False, ), - type="string", - required=False, - ), - "sort": ToolParameter( - description=( - "Sort by column. Format: 'column' or 'column:desc'. " - "Examples: 'store.size:desc' (largest first), 'docs.count:desc', 'index'. " - "Default: 'index' (alphabetical)." + "sort": ToolParameter( + description=( + "Sort by column. Format: 'column' or 'column:desc'. " + "Examples: 'store.size:desc' (largest first), 'docs.count:desc', 'index'. " + "Default: 'index' (alphabetical)." + ), + type="string", + required=False, ), - type="string", - required=False, - ), - "columns": ToolParameter( - description=( - "Comma-separated columns to return. Available: index, health, status, pri, rep, " - "docs.count, docs.deleted, store.size, pri.store.size, creation.date, creation.date.string. " - "Default: 'index,health,status,docs.count,store.size'" + "columns": ToolParameter( + description=( + "Comma-separated columns to return. Available: index, health, status, pri, rep, " + "docs.count, docs.deleted, store.size, pri.store.size, creation.date, creation.date.string. " + "Default: 'index,health,status,docs.count,store.size'" + ), + type="string", + required=False, ), - type="string", - required=False, - ), - "health": ToolParameter( - description="Filter by index health: green, yellow, or red", - type="string", - required=False, - ), - "bytes": ToolParameter( - description="Unit for byte sizes: b, kb, mb, gb, tb, pb. Default: human-readable.", - type="string", - required=False, - ), - "pri": ToolParameter( - description="If true, return only primary shard statistics", - type="boolean", - required=False, - ), - "expand_wildcards": ToolParameter( - description="Which indices to expand wildcards to: open, closed, hidden, none, all. Default: open", - type="string", - required=False, - ), - }), + "health": ToolParameter( + description="Filter by index health: green, yellow, or red", + type="string", + required=False, + ), + "bytes": ToolParameter( + description="Unit for byte sizes: b, kb, mb, gb, tb, pb. Default: human-readable.", + type="string", + required=False, + ), + "pri": ToolParameter( + description="If true, return only primary shard statistics", + type="boolean", + required=False, + ), + "expand_wildcards": ToolParameter( + description="Which indices to expand wildcards to: open, closed, hidden, none, all. Default: open", + type="string", + required=False, + ), + } + ), ) def _invoke(self, params: dict, context: ToolInvokeContext) -> StructuredToolResult: diff --git a/holmes/plugins/toolsets/grafana/common.py b/holmes/plugins/toolsets/grafana/common.py index ab90b309f1..0586befd5f 100644 --- a/holmes/plugins/toolsets/grafana/common.py +++ b/holmes/plugins/toolsets/grafana/common.py @@ -1,11 +1,7 @@ -import datetime -import json from typing import Dict, Optional from pydantic import BaseModel -from holmes.core.tools import StructuredToolResult, StructuredToolResultStatus - class GrafanaConfig(BaseModel): """A config that represents one of the Grafana related tools like Loki or Tempo diff --git a/tests/generate_eval_report.py b/tests/generate_eval_report.py index 3cf57b22ef..1392e5c4b5 100755 --- a/tests/generate_eval_report.py +++ b/tests/generate_eval_report.py @@ -9,7 +9,7 @@ from typing import Any, DefaultDict, Dict, List, Optional, Set from urllib.parse import quote -from llm.utils.test_env_vars import ( +from tests.llm.utils.test_env_vars import ( BRAINTRUST_API_KEY, BRAINTRUST_ORG, BRAINTRUST_PROJECT, From d0348c3b55c409ccac5d27b172e6bd313324c58c Mon Sep 17 00:00:00 2001 From: Tomer Keshet Date: Mon, 5 Jan 2026 18:04:27 +0200 Subject: [PATCH 6/7] fix import errors in action Signed-off-by: Tomer Keshet --- .github/workflows/eval-benchmarks.yaml | 6 +++--- run_benchmarks_local.py | 9 +++++++-- 2 files changed, 10 insertions(+), 5 deletions(-) diff --git a/.github/workflows/eval-benchmarks.yaml b/.github/workflows/eval-benchmarks.yaml index 04be5f41db..ad4a111637 100644 --- a/.github/workflows/eval-benchmarks.yaml +++ b/.github/workflows/eval-benchmarks.yaml @@ -15,7 +15,7 @@ on: models: description: 'Comma-separated list of models to test' required: false - default: 'gpt-4.1' + default: 'azure/gpt-4.1' test_markers: description: 'Custom pytest markers (ONLY use if not using benchmark_type). Cannot be combined with benchmark_type.' required: false @@ -83,9 +83,9 @@ jobs: CMD="$CMD --iterations $ITERATIONS" fi else - # Scheduled run: use fast-benchmark with gpt-4.1 + # Scheduled run: use fast-benchmark with azure/gpt-4.1 BENCHMARK_TYPE="fast-benchmark" - CMD="$CMD --benchmark-type fast-benchmark --models gpt-4.1" + CMD="$CMD --benchmark-type fast-benchmark --models azure/gpt-4.1" fi echo "command=$CMD" >> $GITHUB_OUTPUT diff --git a/run_benchmarks_local.py b/run_benchmarks_local.py index c5501338ec..2f8123b153 100755 --- a/run_benchmarks_local.py +++ b/run_benchmarks_local.py @@ -239,8 +239,13 @@ def generate_report(self) -> None: if self.benchmark_type: cmd.extend(["--benchmark-type", self.benchmark_type]) + # Set up environment with PYTHONPATH to ensure tests module is importable + env = os.environ.copy() + project_root = str(Path.cwd()) + env["PYTHONPATH"] = f"{project_root}:{env.get('PYTHONPATH', '')}" + try: - subprocess.run(cmd, check=True) + subprocess.run(cmd, check=True, env=env) print(f"✅ Report generated: {main_output}") # Generate historical copy first (⚡ in title distinguishes fast benchmarks) @@ -251,7 +256,7 @@ def generate_report(self) -> None: cmd_history = cmd.copy() output_idx = cmd_history.index("--output-file") + 1 cmd_history[output_idx] = str(history_output) - subprocess.run(cmd_history, check=True) + subprocess.run(cmd_history, check=True, env=env) print(f"📁 Saved historical copy: {history_output}") # Create redirect page for latest-results.md pointing to the history file From a9749791f9202fcf97626ad4e29135c4254fb7aa Mon Sep 17 00:00:00 2001 From: Tomer Keshet Date: Tue, 6 Jan 2026 09:06:52 +0200 Subject: [PATCH 7/7] edit existing PR description Signed-off-by: Tomer Keshet --- .github/workflows/eval-benchmarks.yaml | 3 +++ 1 file changed, 3 insertions(+) diff --git a/.github/workflows/eval-benchmarks.yaml b/.github/workflows/eval-benchmarks.yaml index ad4a111637..73ca197433 100644 --- a/.github/workflows/eval-benchmarks.yaml +++ b/.github/workflows/eval-benchmarks.yaml @@ -153,6 +153,9 @@ jobs: EXISTING_PR=$(gh pr list --head "$BRANCH_NAME" --json number --jq '.[0].number') if [ -n "$EXISTING_PR" ]; then echo "Updating existing PR #$EXISTING_PR" + gh pr edit "$EXISTING_PR" \ + --title "$PR_TITLE" \ + --body "$PR_BODY" else gh pr create \ --title "$PR_TITLE" \