diff --git a/.github/scripts/eval-comment-helpers.js b/.github/scripts/eval-comment-helpers.js index 5d00cc8b00..0f874b6219 100644 --- a/.github/scripts/eval-comment-helpers.js +++ b/.github/scripts/eval-comment-helpers.js @@ -26,8 +26,6 @@ function buildParams(raw) { testPreview: raw.test_preview || '', duration: raw.duration || 'N/A', validMarkers: raw.valid_markers || '', - askHolmesEvals: raw.ask_holmes_evals || '', - investigateEvals: raw.investigate_evals || '', triggered_by: raw.triggered_by || '' }; } @@ -46,9 +44,16 @@ function renderProgress(steps) { /** * Render parameters table for manual runs * @param {Object} p - Parameters object + * @param {Object} context - GitHub context object (optional, for rerun link) * @returns {string} Markdown table */ -function renderParamsTable(p) { +function renderParamsTable(p, context = null) { + let workflowLinks = `[View logs](${p.runUrl})`; + if (context) { + const baseWorkflowUrl = `https://github.com/${context.repo.owner}/${context.repo.repo}/actions/workflows/eval-regression.yaml`; + const rerunUrl = p.displayBranch ? `${baseWorkflowUrl}?ref=${encodeURIComponent(p.displayBranch)}` : baseWorkflowUrl; + workflowLinks += ` \\| [Rerun](${rerunUrl})`; + } return `| Parameter | Value |\n|-----------|-------|\n` + `| **Triggered via** | ${p.trigger} |\n` + (p.displayBranch ? `| **Branch** | \`${p.displayBranch}\` |\n` : '') + @@ -57,20 +62,20 @@ function renderParamsTable(p) { (p.filter ? `| **Filter (-k)** | \`${p.filter}\` |\n` : '') + `| **Iterations** | ${p.iterations} |\n` + (p.duration ? `| **Duration** | ${p.duration} |\n` : '') + - `| **Workflow** | [View logs](${p.runUrl}) |\n`; + `| **Workflow** | ${workflowLinks} |\n`; } /** * Build comment body based on state * @param {Object} p - Parameters object * @param {Array<[boolean, string]>} progressSteps - Progress steps (null to hide) - * @param {Object} extras - Extra options (icon, title, testPreview) + * @param {Object} extras - Extra options (icon, title, testPreview, context) * @returns {string} Markdown body */ function buildBody(p, progressSteps, extras = {}) { let body = p.isManual ? `## ${extras.icon || '๐Ÿš€'} ${extras.title || 'Manual Eval Running...'}\n\n` + - renderParamsTable(p) + renderParamsTable(p, extras.context) : `## ${extras.icon || 'โณ'} ${extras.title || 'HolmesGPT evals running...'}\n\n` + `Automatically triggered by ${p.trigger}\n\n` + `[View workflow logs](${p.runUrl})\n`; @@ -87,27 +92,63 @@ function buildBody(p, progressSteps, extras = {}) { return body; } +/** + * Format comma-separated items as code-styled list + * @param {string} items - Comma-separated items + * @returns {string} Formatted items + */ +function formatAsCodes(items) { + if (!items) return '_(loading...)_'; + return items.split(',').map(item => { + const trimmed = item.trim(); + if (!trimmed) return ''; + return `\`${trimmed}\``; + }).filter(Boolean).join(', '); +} + /** * Build re-run instructions footer for automatic runs - * @param {Object} p - Parameters object with validMarkers, askHolmesEvals, investigateEvals + * @param {Object} p - Parameters object with validMarkers * @param {Object} context - GitHub context object + * @param {Object} options - Options (includeLegend: boolean) * @returns {string} Markdown footer */ -function buildRerunFooter(p, context) { - const workflowUrl = `https://github.com/${context.repo.owner}/${context.repo.repo}/actions/workflows/eval-regression.yaml`; - return '\n
\n๐Ÿ“– Legend\n\n' + - '| Icon | Meaning |\n|------|--------|\n' + - '| โœ… | The test was successful |\n' + - '| โž– | The test was skipped |\n' + - '| โš ๏ธ | The test failed but is known to be flaky or known to fail |\n' + - '| ๐Ÿšง | The test had a setup failure (not a code regression) |\n' + - '| ๐Ÿ”ง | The test failed due to mock data issues (not a code regression) |\n' + - '| ๐Ÿšซ | The test was throttled by API rate limits/overload |\n' + - '| โŒ | The test failed and should be fixed before merging the PR |\n' + - '
\n' + - '\n
\n๐Ÿ”„ Re-run evals manually\n\n' + - '> โš ๏ธ **Warning:** Manual re-runs have NO default markers and will run ALL LLM tests (~100+), which can take 1+ hours. ' + - 'Use `markers: regression` or `filter: test_name` to limit scope.\n\n' + +function buildRerunFooter(p, context, options = {}) { + const { includeLegend = false } = options; + const repoFullName = `${context.repo.owner}/${context.repo.repo}`; + const baseWorkflowUrl = `https://github.com/${repoFullName}/actions/workflows/eval-regression.yaml`; + const workflowUrl = p.displayBranch ? `${baseWorkflowUrl}?ref=${encodeURIComponent(p.displayBranch)}` : baseWorkflowUrl; + + // Format markers as comma-separated code-styled names + const markersFormatted = formatAsCodes(p.validMarkers); + + // gh CLI command to run workflow from PR branch + const ghCommand = p.displayBranch + ? `gh workflow run eval-regression.yaml --repo ${repoFullName} --ref ${p.displayBranch} -f markers=regression` + : `gh workflow run eval-regression.yaml --repo ${repoFullName} -f markers=regression`; + + let footer = ''; + + // Only show legend when results are displayed + if (includeLegend) { + footer += '\n
\n๐Ÿ“– Legend\n\n' + + '| Icon | Meaning |\n|------|--------|\n' + + '| โœ… | The test was successful |\n' + + '| โž– | The test was skipped |\n' + + '| โš ๏ธ | The test failed but is known to be flaky or known to fail |\n' + + '| ๐Ÿšง | The test had a setup failure (not a code regression) |\n' + + '| ๐Ÿ”ง | The test failed due to mock data issues (not a code regression) |\n' + + '| ๐Ÿšซ | The test was throttled by API rate limits/overload |\n' + + '| โŒ | The test failed and should be fixed before merging the PR |\n' + + '
\n'; + } + + footer += '\n
\n๐Ÿ”„ Re-run evals manually\n\n' + + '> โš ๏ธ **Warning:** `/eval` comments always run using the **workflow from master**, not from this PR branch. ' + + 'If you modified the GitHub Action (e.g., added secrets or env vars), those changes won\'t take effect.\n>\n' + + '> **To test workflow changes**, use the GitHub CLI or [Actions UI](' + workflowUrl + ') instead:\n>\n' + + '> ```\n> ' + ghCommand + '\n> ```\n\n' + + '---\n\n' + '**Option 1: Comment on this PR** with `/eval`:\n\n' + '```\n/eval\nmarkers: regression\n```\n\n' + 'Or with more options (one per line):\n\n' + @@ -117,20 +158,17 @@ function buildRerunFooter(p, context) { '| Option | Description |\n|--------|-------------|\n' + '| `model` | Model(s) to test (default: same as automatic runs) |\n' + '| `markers` | Pytest markers (**no default - runs all tests!**) |\n' + - '| `filter` | Pytest -k filter |\n' + + '| `filter` | Pytest -k filter (use `/list` to see valid eval names) |\n' + '| `iterations` | Number of runs, max 10 |\n' + '| `branch` | Run evals on a different branch (for cross-branch comparison) |\n\n' + '**Quick re-run:** Use `/last` to re-run the most recent `/eval` on this PR with the same parameters.\n\n' + `**Option 2: [Trigger via GitHub Actions UI](${workflowUrl})** โ†’ "Run workflow"\n
\n` + '\n
\n๐Ÿท๏ธ Valid markers\n\n' + - (p.validMarkers || '_(Collecting from pyproject.toml...)_') + + markersFormatted + '\n
\n' + - '\n
\n๐Ÿ“‹ Valid eval names (use with filter)\n\n' + - '**test_ask_holmes:**\n' + - (p.askHolmesEvals || '_(Collecting from tests/llm/fixtures/...)_') + - '\n\n**test_investigate:**\n' + - (p.investigateEvals || '_(Collecting from tests/llm/fixtures/...)_') + - '\n
\n'; + '\n---\n**Commands:** `/eval` ยท `/last` ยท `/list`\n'; + + return footer; } module.exports = { diff --git a/.github/workflows/eval-regression.yaml b/.github/workflows/eval-regression.yaml index 1652db48d5..bf600937b5 100644 --- a/.github/workflows/eval-regression.yaml +++ b/.github/workflows/eval-regression.yaml @@ -36,6 +36,61 @@ permissions: issues: write jobs: + # Handle /list command - outputs available eval names + list_evals: + runs-on: ubuntu-latest + if: github.event_name == 'issue_comment' && github.event.issue.pull_request && startsWith(github.event.comment.body, '/list') + steps: + - name: Check permissions + uses: actions/github-script@v7 + with: + script: | + const association = context.payload.comment.author_association; + if (!['OWNER', 'MEMBER', 'COLLABORATOR'].includes(association)) { + core.setFailed(`Permission denied: ${association} cannot use /list (requires OWNER, MEMBER, or COLLABORATOR)`); + } + + - name: Checkout code + uses: actions/checkout@v4 + + - name: Post eval list + uses: actions/github-script@v7 + with: + script: | + const fs = require('fs'); + const path = require('path'); + + // Collect eval names from fixture directories + const askHolmesPath = 'tests/llm/fixtures/test_ask_holmes'; + const investigatePath = 'tests/llm/fixtures/test_investigate'; + + const getEvalNames = (dir) => { + try { + return fs.readdirSync(dir) + .filter(f => !f.startsWith('.')) + .sort() + .map(f => `\`${f}\``) + .join(', '); + } catch (e) { + return '_(none found)_'; + } + }; + + const askHolmesEvals = getEvalNames(askHolmesPath); + const investigateEvals = getEvalNames(investigatePath); + + const body = `## ๐Ÿ“‹ Available Eval Names\n\n` + + `Use these with \`filter:\` in your \`/eval\` command.\n\n` + + `**test_ask_holmes:**\n${askHolmesEvals}\n\n` + + `**test_investigate:**\n${investigateEvals}`; + + await github.rest.issues.createComment({ + owner: context.repo.owner, + repo: context.repo.repo, + issue_number: context.issue.number, + body: body + }); + llm_evals: runs-on: ubuntu-latest if: | @@ -335,7 +390,7 @@ jobs: owner: context.repo.owner, repo: context.repo.repo, issue_number: p.prNumber, - body: buildBody(p, progressSteps) + buildRerunFooter(p, context) + body: buildBody(p, progressSteps, { context }) + buildRerunFooter(p, context) }); core.setOutput('comment_id', comment.data.id.toString()); @@ -367,7 +422,7 @@ jobs: owner: context.repo.owner, repo: context.repo.repo, comment_id: p.commentId, - body: buildBody(p, progressSteps) + buildRerunFooter(p, context) + body: buildBody(p, progressSteps, { context }) + buildRerunFooter(p, context) }); - name: Collect evals to run @@ -390,22 +445,17 @@ jobs: TEST_COUNT=$(echo "$TEST_LIST" | grep -c "^tests/llm/" || true) # Collect valid markers from pyproject.toml using Python/tomllib (excluding 'llm' which is internal) + # Output as comma-separated plain names VALID_MARKERS=$(python3 << 'PYEOF' import tomllib with open('pyproject.toml', 'rb') as f: config = tomllib.load(f) markers = config.get('tool', {}).get('pytest', {}).get('ini_options', {}).get('markers', []) - for m in sorted(markers): - name = m.split(':')[0] - if name != 'llm': - print(f'- `{name}`') + names = [m.split(':')[0] for m in sorted(markers) if m.split(':')[0] != 'llm'] + print(','.join(names)) PYEOF ) - # Collect valid eval names from fixture directories - ASK_HOLMES_EVALS=$(ls -1 tests/llm/fixtures/test_ask_holmes/ 2>/dev/null | grep -v '^\.' | sort | awk '{print "- \x60" $0 "\x60"}') - INVESTIGATE_EVALS=$(ls -1 tests/llm/fixtures/test_investigate/ 2>/dev/null | grep -v '^\.' | sort | awk '{print "- \x60" $0 "\x60"}') - # Output for next steps echo "test_count=$TEST_COUNT" >> $GITHUB_OUTPUT @@ -419,16 +469,6 @@ jobs: echo "$VALID_MARKERS" >> $GITHUB_OUTPUT echo "$EOF" >> $GITHUB_OUTPUT - EOF=$(dd if=/dev/urandom bs=15 count=1 status=none | base64) - echo "ask_holmes_evals<<$EOF" >> $GITHUB_OUTPUT - echo "$ASK_HOLMES_EVALS" >> $GITHUB_OUTPUT - echo "$EOF" >> $GITHUB_OUTPUT - - EOF=$(dd if=/dev/urandom bs=15 count=1 status=none | base64) - echo "investigate_evals<<$EOF" >> $GITHUB_OUTPUT - echo "$INVESTIGATE_EVALS" >> $GITHUB_OUTPUT - echo "$EOF" >> $GITHUB_OUTPUT - - name: Update progress - evals collected if: steps.check-tests.outputs.should-run == 'true' && steps.eval-params.outputs.pr_number != '' && steps.initial-comment.outputs.comment_id != '' uses: actions/github-script@v7 @@ -440,9 +480,7 @@ jobs: comment_id: ${{ toJSON(steps.initial-comment.outputs.comment_id) }}, test_count: ${{ toJSON(steps.test-preview.outputs.test_count) }}, test_preview: ${{ toJSON(steps.test-preview.outputs.test_preview) }}, - valid_markers: ${{ toJSON(steps.test-preview.outputs.valid_markers) }}, - ask_holmes_evals: ${{ toJSON(steps.test-preview.outputs.ask_holmes_evals) }}, - investigate_evals: ${{ toJSON(steps.test-preview.outputs.investigate_evals) }} + valid_markers: ${{ toJSON(steps.test-preview.outputs.valid_markers) }} }); const progressSteps = [ [true, 'Setup HolmesGPT environment'], @@ -454,7 +492,7 @@ jobs: owner: context.repo.owner, repo: context.repo.repo, comment_id: p.commentId, - body: buildBody(p, progressSteps, { testPreview: p.testCount !== '0' ? p.testPreview : null }) + buildRerunFooter(p, context) + body: buildBody(p, progressSteps, { testPreview: p.testCount !== '0' ? p.testPreview : null, context }) + buildRerunFooter(p, context) }); - name: Setup KIND cluster @@ -475,9 +513,7 @@ jobs: comment_id: ${{ toJSON(steps.initial-comment.outputs.comment_id) }}, test_count: ${{ toJSON(steps.test-preview.outputs.test_count) }}, test_preview: ${{ toJSON(steps.test-preview.outputs.test_preview) }}, - valid_markers: ${{ toJSON(steps.test-preview.outputs.valid_markers) }}, - ask_holmes_evals: ${{ toJSON(steps.test-preview.outputs.ask_holmes_evals) }}, - investigate_evals: ${{ toJSON(steps.test-preview.outputs.investigate_evals) }} + valid_markers: ${{ toJSON(steps.test-preview.outputs.valid_markers) }} }); const progressSteps = [ [true, 'Setup HolmesGPT environment'], @@ -489,7 +525,7 @@ jobs: owner: context.repo.owner, repo: context.repo.repo, comment_id: p.commentId, - body: buildBody(p, progressSteps, { testPreview: p.testCount !== '0' ? p.testPreview : null }) + buildRerunFooter(p, context) + body: buildBody(p, progressSteps, { testPreview: p.testCount !== '0' ? p.testPreview : null, context }) + buildRerunFooter(p, context) }); - name: Run tests @@ -547,8 +583,6 @@ jobs: comment_id: ${{ toJSON(steps.initial-comment.outputs.comment_id) }}, duration: ${{ toJSON(steps.evals.outputs.duration) }}, valid_markers: ${{ toJSON(steps.test-preview.outputs.valid_markers) }}, - ask_holmes_evals: ${{ toJSON(steps.test-preview.outputs.ask_holmes_evals) }}, - investigate_evals: ${{ toJSON(steps.test-preview.outputs.investigate_evals) }}, triggered_by: ${{ toJSON(steps.eval-params.outputs.triggered_by) }} }); const report = fs.existsSync('evals_report.md') ? fs.readFileSync('evals_report.md', 'utf8') : ''; @@ -564,11 +598,12 @@ jobs: let body = notifyHeader + buildBody(p, null, { icon: p.isManual ? '๐Ÿงช' : 'โœ…', - title: p.isManual ? 'Manual Eval Results' : 'Results of HolmesGPT evals' + title: p.isManual ? 'Manual Eval Results' : 'Results of HolmesGPT evals', + context }); body += '\n' + (report || 'โš ๏ธ No eval report was generated.\n\n'); if (hasFailures) body += `\n### โš ๏ธ ${failures} Failure${failures === '1' ? '' : 's'} Detected\n\n`; - body += buildRerunFooter(p, context); + body += buildRerunFooter(p, context, { includeLegend: true }); // Delete progress comment and create fresh results comment // This ensures the user gets a GitHub notification when evals complete diff --git a/tests/llm/utils/reporting/github_reporter.py b/tests/llm/utils/reporting/github_reporter.py index 49871effc1..620f30b7bd 100644 --- a/tests/llm/utils/reporting/github_reporter.py +++ b/tests/llm/utils/reporting/github_reporter.py @@ -64,7 +64,7 @@ def _generate_historical_details_section(details: HistoricalComparisonDetails) - Returns: Markdown string with collapsible details section """ - lines = ["
", "Historical Comparison Details\n"] + lines = ["
", "Historical Comparison Details\n"] # Filter description if details.filter_description: