diff --git a/.github/scripts/eval-comment-helpers.js b/.github/scripts/eval-comment-helpers.js
index 5d00cc8b00..0f874b6219 100644
--- a/.github/scripts/eval-comment-helpers.js
+++ b/.github/scripts/eval-comment-helpers.js
@@ -26,8 +26,6 @@ function buildParams(raw) {
testPreview: raw.test_preview || '',
duration: raw.duration || 'N/A',
validMarkers: raw.valid_markers || '',
- askHolmesEvals: raw.ask_holmes_evals || '',
- investigateEvals: raw.investigate_evals || '',
triggered_by: raw.triggered_by || ''
};
}
@@ -46,9 +44,16 @@ function renderProgress(steps) {
/**
* Render parameters table for manual runs
* @param {Object} p - Parameters object
+ * @param {Object} context - GitHub context object (optional, for rerun link)
* @returns {string} Markdown table
*/
-function renderParamsTable(p) {
+function renderParamsTable(p, context = null) {
+ let workflowLinks = `[View logs](${p.runUrl})`;
+ if (context) {
+ const baseWorkflowUrl = `https://github.com/${context.repo.owner}/${context.repo.repo}/actions/workflows/eval-regression.yaml`;
+ const rerunUrl = p.displayBranch ? `${baseWorkflowUrl}?ref=${encodeURIComponent(p.displayBranch)}` : baseWorkflowUrl;
+ workflowLinks += ` \\| [Rerun](${rerunUrl})`;
+ }
return `| Parameter | Value |\n|-----------|-------|\n` +
`| **Triggered via** | ${p.trigger} |\n` +
(p.displayBranch ? `| **Branch** | \`${p.displayBranch}\` |\n` : '') +
@@ -57,20 +62,20 @@ function renderParamsTable(p) {
(p.filter ? `| **Filter (-k)** | \`${p.filter}\` |\n` : '') +
`| **Iterations** | ${p.iterations} |\n` +
(p.duration ? `| **Duration** | ${p.duration} |\n` : '') +
- `| **Workflow** | [View logs](${p.runUrl}) |\n`;
+ `| **Workflow** | ${workflowLinks} |\n`;
}
/**
* Build comment body based on state
* @param {Object} p - Parameters object
* @param {Array<[boolean, string]>} progressSteps - Progress steps (null to hide)
- * @param {Object} extras - Extra options (icon, title, testPreview)
+ * @param {Object} extras - Extra options (icon, title, testPreview, context)
* @returns {string} Markdown body
*/
function buildBody(p, progressSteps, extras = {}) {
let body = p.isManual
? `## ${extras.icon || '๐'} ${extras.title || 'Manual Eval Running...'}\n\n` +
- renderParamsTable(p)
+ renderParamsTable(p, extras.context)
: `## ${extras.icon || 'โณ'} ${extras.title || 'HolmesGPT evals running...'}\n\n` +
`Automatically triggered by ${p.trigger}\n\n` +
`[View workflow logs](${p.runUrl})\n`;
@@ -87,27 +92,63 @@ function buildBody(p, progressSteps, extras = {}) {
return body;
}
+/**
+ * Format comma-separated items as code-styled list
+ * @param {string} items - Comma-separated items
+ * @returns {string} Formatted items
+ */
+function formatAsCodes(items) {
+ if (!items) return '_(loading...)_';
+ return items.split(',').map(item => {
+ const trimmed = item.trim();
+ if (!trimmed) return '';
+ return `\`${trimmed}\``;
+ }).filter(Boolean).join(', ');
+}
+
/**
* Build re-run instructions footer for automatic runs
- * @param {Object} p - Parameters object with validMarkers, askHolmesEvals, investigateEvals
+ * @param {Object} p - Parameters object with validMarkers
* @param {Object} context - GitHub context object
+ * @param {Object} options - Options (includeLegend: boolean)
* @returns {string} Markdown footer
*/
-function buildRerunFooter(p, context) {
- const workflowUrl = `https://github.com/${context.repo.owner}/${context.repo.repo}/actions/workflows/eval-regression.yaml`;
- return '\n\n๐ Legend
\n\n' +
- '| Icon | Meaning |\n|------|--------|\n' +
- '| โ
| The test was successful |\n' +
- '| โ | The test was skipped |\n' +
- '| โ ๏ธ | The test failed but is known to be flaky or known to fail |\n' +
- '| ๐ง | The test had a setup failure (not a code regression) |\n' +
- '| ๐ง | The test failed due to mock data issues (not a code regression) |\n' +
- '| ๐ซ | The test was throttled by API rate limits/overload |\n' +
- '| โ | The test failed and should be fixed before merging the PR |\n' +
- ' \n' +
- '\n\n๐ Re-run evals manually
\n\n' +
- '> โ ๏ธ **Warning:** Manual re-runs have NO default markers and will run ALL LLM tests (~100+), which can take 1+ hours. ' +
- 'Use `markers: regression` or `filter: test_name` to limit scope.\n\n' +
+function buildRerunFooter(p, context, options = {}) {
+ const { includeLegend = false } = options;
+ const repoFullName = `${context.repo.owner}/${context.repo.repo}`;
+ const baseWorkflowUrl = `https://github.com/${repoFullName}/actions/workflows/eval-regression.yaml`;
+ const workflowUrl = p.displayBranch ? `${baseWorkflowUrl}?ref=${encodeURIComponent(p.displayBranch)}` : baseWorkflowUrl;
+
+ // Format markers as comma-separated code-styled names
+ const markersFormatted = formatAsCodes(p.validMarkers);
+
+ // gh CLI command to run workflow from PR branch
+ const ghCommand = p.displayBranch
+ ? `gh workflow run eval-regression.yaml --repo ${repoFullName} --ref ${p.displayBranch} -f markers=regression`
+ : `gh workflow run eval-regression.yaml --repo ${repoFullName} -f markers=regression`;
+
+ let footer = '';
+
+ // Only show legend when results are displayed
+ if (includeLegend) {
+ footer += '\n\n๐ Legend
\n\n' +
+ '| Icon | Meaning |\n|------|--------|\n' +
+ '| โ
| The test was successful |\n' +
+ '| โ | The test was skipped |\n' +
+ '| โ ๏ธ | The test failed but is known to be flaky or known to fail |\n' +
+ '| ๐ง | The test had a setup failure (not a code regression) |\n' +
+ '| ๐ง | The test failed due to mock data issues (not a code regression) |\n' +
+ '| ๐ซ | The test was throttled by API rate limits/overload |\n' +
+ '| โ | The test failed and should be fixed before merging the PR |\n' +
+ ' \n';
+ }
+
+ footer += '\n\n๐ Re-run evals manually
\n\n' +
+ '> โ ๏ธ **Warning:** `/eval` comments always run using the **workflow from master**, not from this PR branch. ' +
+ 'If you modified the GitHub Action (e.g., added secrets or env vars), those changes won\'t take effect.\n>\n' +
+ '> **To test workflow changes**, use the GitHub CLI or [Actions UI](' + workflowUrl + ') instead:\n>\n' +
+ '> ```\n> ' + ghCommand + '\n> ```\n\n' +
+ '---\n\n' +
'**Option 1: Comment on this PR** with `/eval`:\n\n' +
'```\n/eval\nmarkers: regression\n```\n\n' +
'Or with more options (one per line):\n\n' +
@@ -117,20 +158,17 @@ function buildRerunFooter(p, context) {
'| Option | Description |\n|--------|-------------|\n' +
'| `model` | Model(s) to test (default: same as automatic runs) |\n' +
'| `markers` | Pytest markers (**no default - runs all tests!**) |\n' +
- '| `filter` | Pytest -k filter |\n' +
+ '| `filter` | Pytest -k filter (use `/list` to see valid eval names) |\n' +
'| `iterations` | Number of runs, max 10 |\n' +
'| `branch` | Run evals on a different branch (for cross-branch comparison) |\n\n' +
'**Quick re-run:** Use `/last` to re-run the most recent `/eval` on this PR with the same parameters.\n\n' +
`**Option 2: [Trigger via GitHub Actions UI](${workflowUrl})** โ "Run workflow"\n \n` +
'\n\n๐ท๏ธ Valid markers
\n\n' +
- (p.validMarkers || '_(Collecting from pyproject.toml...)_') +
+ markersFormatted +
'\n \n' +
- '\n\n๐ Valid eval names (use with filter)
\n\n' +
- '**test_ask_holmes:**\n' +
- (p.askHolmesEvals || '_(Collecting from tests/llm/fixtures/...)_') +
- '\n\n**test_investigate:**\n' +
- (p.investigateEvals || '_(Collecting from tests/llm/fixtures/...)_') +
- '\n \n';
+ '\n---\n**Commands:** `/eval` ยท `/last` ยท `/list`\n';
+
+ return footer;
}
module.exports = {
diff --git a/.github/workflows/eval-regression.yaml b/.github/workflows/eval-regression.yaml
index 1652db48d5..bf600937b5 100644
--- a/.github/workflows/eval-regression.yaml
+++ b/.github/workflows/eval-regression.yaml
@@ -36,6 +36,61 @@ permissions:
issues: write
jobs:
+ # Handle /list command - outputs available eval names
+ list_evals:
+ runs-on: ubuntu-latest
+ if: github.event_name == 'issue_comment' && github.event.issue.pull_request && startsWith(github.event.comment.body, '/list')
+ steps:
+ - name: Check permissions
+ uses: actions/github-script@v7
+ with:
+ script: |
+ const association = context.payload.comment.author_association;
+ if (!['OWNER', 'MEMBER', 'COLLABORATOR'].includes(association)) {
+ core.setFailed(`Permission denied: ${association} cannot use /list (requires OWNER, MEMBER, or COLLABORATOR)`);
+ }
+
+ - name: Checkout code
+ uses: actions/checkout@v4
+
+ - name: Post eval list
+ uses: actions/github-script@v7
+ with:
+ script: |
+ const fs = require('fs');
+ const path = require('path');
+
+ // Collect eval names from fixture directories
+ const askHolmesPath = 'tests/llm/fixtures/test_ask_holmes';
+ const investigatePath = 'tests/llm/fixtures/test_investigate';
+
+ const getEvalNames = (dir) => {
+ try {
+ return fs.readdirSync(dir)
+ .filter(f => !f.startsWith('.'))
+ .sort()
+ .map(f => `\`${f}\``)
+ .join(', ');
+ } catch (e) {
+ return '_(none found)_';
+ }
+ };
+
+ const askHolmesEvals = getEvalNames(askHolmesPath);
+ const investigateEvals = getEvalNames(investigatePath);
+
+ const body = `## ๐ Available Eval Names\n\n` +
+ `Use these with \`filter:\` in your \`/eval\` command.\n\n` +
+ `**test_ask_holmes:**\n${askHolmesEvals}\n\n` +
+ `**test_investigate:**\n${investigateEvals}`;
+
+ await github.rest.issues.createComment({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ issue_number: context.issue.number,
+ body: body
+ });
+
llm_evals:
runs-on: ubuntu-latest
if: |
@@ -335,7 +390,7 @@ jobs:
owner: context.repo.owner,
repo: context.repo.repo,
issue_number: p.prNumber,
- body: buildBody(p, progressSteps) + buildRerunFooter(p, context)
+ body: buildBody(p, progressSteps, { context }) + buildRerunFooter(p, context)
});
core.setOutput('comment_id', comment.data.id.toString());
@@ -367,7 +422,7 @@ jobs:
owner: context.repo.owner,
repo: context.repo.repo,
comment_id: p.commentId,
- body: buildBody(p, progressSteps) + buildRerunFooter(p, context)
+ body: buildBody(p, progressSteps, { context }) + buildRerunFooter(p, context)
});
- name: Collect evals to run
@@ -390,22 +445,17 @@ jobs:
TEST_COUNT=$(echo "$TEST_LIST" | grep -c "^tests/llm/" || true)
# Collect valid markers from pyproject.toml using Python/tomllib (excluding 'llm' which is internal)
+ # Output as comma-separated plain names
VALID_MARKERS=$(python3 << 'PYEOF'
import tomllib
with open('pyproject.toml', 'rb') as f:
config = tomllib.load(f)
markers = config.get('tool', {}).get('pytest', {}).get('ini_options', {}).get('markers', [])
- for m in sorted(markers):
- name = m.split(':')[0]
- if name != 'llm':
- print(f'- `{name}`')
+ names = [m.split(':')[0] for m in sorted(markers) if m.split(':')[0] != 'llm']
+ print(','.join(names))
PYEOF
)
- # Collect valid eval names from fixture directories
- ASK_HOLMES_EVALS=$(ls -1 tests/llm/fixtures/test_ask_holmes/ 2>/dev/null | grep -v '^\.' | sort | awk '{print "- \x60" $0 "\x60"}')
- INVESTIGATE_EVALS=$(ls -1 tests/llm/fixtures/test_investigate/ 2>/dev/null | grep -v '^\.' | sort | awk '{print "- \x60" $0 "\x60"}')
-
# Output for next steps
echo "test_count=$TEST_COUNT" >> $GITHUB_OUTPUT
@@ -419,16 +469,6 @@ jobs:
echo "$VALID_MARKERS" >> $GITHUB_OUTPUT
echo "$EOF" >> $GITHUB_OUTPUT
- EOF=$(dd if=/dev/urandom bs=15 count=1 status=none | base64)
- echo "ask_holmes_evals<<$EOF" >> $GITHUB_OUTPUT
- echo "$ASK_HOLMES_EVALS" >> $GITHUB_OUTPUT
- echo "$EOF" >> $GITHUB_OUTPUT
-
- EOF=$(dd if=/dev/urandom bs=15 count=1 status=none | base64)
- echo "investigate_evals<<$EOF" >> $GITHUB_OUTPUT
- echo "$INVESTIGATE_EVALS" >> $GITHUB_OUTPUT
- echo "$EOF" >> $GITHUB_OUTPUT
-
- name: Update progress - evals collected
if: steps.check-tests.outputs.should-run == 'true' && steps.eval-params.outputs.pr_number != '' && steps.initial-comment.outputs.comment_id != ''
uses: actions/github-script@v7
@@ -440,9 +480,7 @@ jobs:
comment_id: ${{ toJSON(steps.initial-comment.outputs.comment_id) }},
test_count: ${{ toJSON(steps.test-preview.outputs.test_count) }},
test_preview: ${{ toJSON(steps.test-preview.outputs.test_preview) }},
- valid_markers: ${{ toJSON(steps.test-preview.outputs.valid_markers) }},
- ask_holmes_evals: ${{ toJSON(steps.test-preview.outputs.ask_holmes_evals) }},
- investigate_evals: ${{ toJSON(steps.test-preview.outputs.investigate_evals) }}
+ valid_markers: ${{ toJSON(steps.test-preview.outputs.valid_markers) }}
});
const progressSteps = [
[true, 'Setup HolmesGPT environment'],
@@ -454,7 +492,7 @@ jobs:
owner: context.repo.owner,
repo: context.repo.repo,
comment_id: p.commentId,
- body: buildBody(p, progressSteps, { testPreview: p.testCount !== '0' ? p.testPreview : null }) + buildRerunFooter(p, context)
+ body: buildBody(p, progressSteps, { testPreview: p.testCount !== '0' ? p.testPreview : null, context }) + buildRerunFooter(p, context)
});
- name: Setup KIND cluster
@@ -475,9 +513,7 @@ jobs:
comment_id: ${{ toJSON(steps.initial-comment.outputs.comment_id) }},
test_count: ${{ toJSON(steps.test-preview.outputs.test_count) }},
test_preview: ${{ toJSON(steps.test-preview.outputs.test_preview) }},
- valid_markers: ${{ toJSON(steps.test-preview.outputs.valid_markers) }},
- ask_holmes_evals: ${{ toJSON(steps.test-preview.outputs.ask_holmes_evals) }},
- investigate_evals: ${{ toJSON(steps.test-preview.outputs.investigate_evals) }}
+ valid_markers: ${{ toJSON(steps.test-preview.outputs.valid_markers) }}
});
const progressSteps = [
[true, 'Setup HolmesGPT environment'],
@@ -489,7 +525,7 @@ jobs:
owner: context.repo.owner,
repo: context.repo.repo,
comment_id: p.commentId,
- body: buildBody(p, progressSteps, { testPreview: p.testCount !== '0' ? p.testPreview : null }) + buildRerunFooter(p, context)
+ body: buildBody(p, progressSteps, { testPreview: p.testCount !== '0' ? p.testPreview : null, context }) + buildRerunFooter(p, context)
});
- name: Run tests
@@ -547,8 +583,6 @@ jobs:
comment_id: ${{ toJSON(steps.initial-comment.outputs.comment_id) }},
duration: ${{ toJSON(steps.evals.outputs.duration) }},
valid_markers: ${{ toJSON(steps.test-preview.outputs.valid_markers) }},
- ask_holmes_evals: ${{ toJSON(steps.test-preview.outputs.ask_holmes_evals) }},
- investigate_evals: ${{ toJSON(steps.test-preview.outputs.investigate_evals) }},
triggered_by: ${{ toJSON(steps.eval-params.outputs.triggered_by) }}
});
const report = fs.existsSync('evals_report.md') ? fs.readFileSync('evals_report.md', 'utf8') : '';
@@ -564,11 +598,12 @@ jobs:
let body = notifyHeader + buildBody(p, null, {
icon: p.isManual ? '๐งช' : 'โ
',
- title: p.isManual ? 'Manual Eval Results' : 'Results of HolmesGPT evals'
+ title: p.isManual ? 'Manual Eval Results' : 'Results of HolmesGPT evals',
+ context
});
body += '\n' + (report || 'โ ๏ธ No eval report was generated.\n\n');
if (hasFailures) body += `\n### โ ๏ธ ${failures} Failure${failures === '1' ? '' : 's'} Detected\n\n`;
- body += buildRerunFooter(p, context);
+ body += buildRerunFooter(p, context, { includeLegend: true });
// Delete progress comment and create fresh results comment
// This ensures the user gets a GitHub notification when evals complete
diff --git a/tests/llm/utils/reporting/github_reporter.py b/tests/llm/utils/reporting/github_reporter.py
index 49871effc1..620f30b7bd 100644
--- a/tests/llm/utils/reporting/github_reporter.py
+++ b/tests/llm/utils/reporting/github_reporter.py
@@ -64,7 +64,7 @@ def _generate_historical_details_section(details: HistoricalComparisonDetails) -
Returns:
Markdown string with collapsible details section
"""
- lines = ["", "Historical Comparison Details
\n"]
+ lines = ["", "Historical Comparison Details
\n"]
# Filter description
if details.filter_description: