Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
98 changes: 68 additions & 30 deletions .github/scripts/eval-comment-helpers.js
Original file line number Diff line number Diff line change
Expand Up @@ -26,8 +26,6 @@ function buildParams(raw) {
testPreview: raw.test_preview || '',
duration: raw.duration || 'N/A',
validMarkers: raw.valid_markers || '',
askHolmesEvals: raw.ask_holmes_evals || '',
investigateEvals: raw.investigate_evals || '',
triggered_by: raw.triggered_by || ''
};
}
Expand All @@ -46,9 +44,16 @@ function renderProgress(steps) {
/**
* Render parameters table for manual runs
* @param {Object} p - Parameters object
* @param {Object} context - GitHub context object (optional, for rerun link)
* @returns {string} Markdown table
*/
function renderParamsTable(p) {
function renderParamsTable(p, context = null) {
let workflowLinks = `[View logs](${p.runUrl})`;
if (context) {
const baseWorkflowUrl = `https://github.com/${context.repo.owner}/${context.repo.repo}/actions/workflows/eval-regression.yaml`;
const rerunUrl = p.displayBranch ? `${baseWorkflowUrl}?ref=${encodeURIComponent(p.displayBranch)}` : baseWorkflowUrl;
workflowLinks += ` \\| [Rerun](${rerunUrl})`;
}
return `| Parameter | Value |\n|-----------|-------|\n` +
`| **Triggered via** | ${p.trigger} |\n` +
(p.displayBranch ? `| **Branch** | \`${p.displayBranch}\` |\n` : '') +
Expand All @@ -57,20 +62,20 @@ function renderParamsTable(p) {
(p.filter ? `| **Filter (-k)** | \`${p.filter}\` |\n` : '') +
`| **Iterations** | ${p.iterations} |\n` +
(p.duration ? `| **Duration** | ${p.duration} |\n` : '') +
`| **Workflow** | [View logs](${p.runUrl}) |\n`;
`| **Workflow** | ${workflowLinks} |\n`;
}

/**
* Build comment body based on state
* @param {Object} p - Parameters object
* @param {Array<[boolean, string]>} progressSteps - Progress steps (null to hide)
* @param {Object} extras - Extra options (icon, title, testPreview)
* @param {Object} extras - Extra options (icon, title, testPreview, context)
* @returns {string} Markdown body
*/
function buildBody(p, progressSteps, extras = {}) {
let body = p.isManual
? `## ${extras.icon || '🚀'} ${extras.title || 'Manual Eval Running...'}\n\n` +
renderParamsTable(p)
renderParamsTable(p, extras.context)
: `## ${extras.icon || '⏳'} ${extras.title || 'HolmesGPT evals running...'}\n\n` +
`Automatically triggered by ${p.trigger}\n\n` +
`[View workflow logs](${p.runUrl})\n`;
Expand All @@ -87,27 +92,63 @@ function buildBody(p, progressSteps, extras = {}) {
return body;
}

/**
* Format comma-separated items as code-styled list
* @param {string} items - Comma-separated items
* @returns {string} Formatted items
*/
function formatAsCodes(items) {
if (!items) return '_(loading...)_';
return items.split(',').map(item => {
const trimmed = item.trim();
if (!trimmed) return '';
return `\`${trimmed}\``;
}).filter(Boolean).join(', ');
}
Comment thread
aantn marked this conversation as resolved.

/**
* Build re-run instructions footer for automatic runs
* @param {Object} p - Parameters object with validMarkers, askHolmesEvals, investigateEvals
* @param {Object} p - Parameters object with validMarkers
* @param {Object} context - GitHub context object
* @param {Object} options - Options (includeLegend: boolean)
* @returns {string} Markdown footer
*/
function buildRerunFooter(p, context) {
const workflowUrl = `https://github.com/${context.repo.owner}/${context.repo.repo}/actions/workflows/eval-regression.yaml`;
return '\n<details>\n<summary>📖 <b>Legend</b></summary>\n\n' +
'| Icon | Meaning |\n|------|--------|\n' +
'| ✅ | The test was successful |\n' +
'| ➖ | The test was skipped |\n' +
'| ⚠️ | The test failed but is known to be flaky or known to fail |\n' +
'| 🚧 | The test had a setup failure (not a code regression) |\n' +
'| 🔧 | The test failed due to mock data issues (not a code regression) |\n' +
'| 🚫 | The test was throttled by API rate limits/overload |\n' +
'| ❌ | The test failed and should be fixed before merging the PR |\n' +
'</details>\n' +
'\n<details>\n<summary>🔄 <b>Re-run evals manually</b></summary>\n\n' +
'> ⚠️ **Warning:** Manual re-runs have NO default markers and will run ALL LLM tests (~100+), which can take 1+ hours. ' +
'Use `markers: regression` or `filter: test_name` to limit scope.\n\n' +
function buildRerunFooter(p, context, options = {}) {
const { includeLegend = false } = options;
const repoFullName = `${context.repo.owner}/${context.repo.repo}`;
const baseWorkflowUrl = `https://github.com/${repoFullName}/actions/workflows/eval-regression.yaml`;
const workflowUrl = p.displayBranch ? `${baseWorkflowUrl}?ref=${encodeURIComponent(p.displayBranch)}` : baseWorkflowUrl;

// Format markers as comma-separated code-styled names
const markersFormatted = formatAsCodes(p.validMarkers);

// gh CLI command to run workflow from PR branch
const ghCommand = p.displayBranch
? `gh workflow run eval-regression.yaml --repo ${repoFullName} --ref ${p.displayBranch} -f markers=regression`
: `gh workflow run eval-regression.yaml --repo ${repoFullName} -f markers=regression`;

let footer = '';

// Only show legend when results are displayed
if (includeLegend) {
footer += '\n<details>\n<summary>📖 <b>Legend</b></summary>\n\n' +
'| Icon | Meaning |\n|------|--------|\n' +
'| ✅ | The test was successful |\n' +
'| ➖ | The test was skipped |\n' +
'| ⚠️ | The test failed but is known to be flaky or known to fail |\n' +
'| 🚧 | The test had a setup failure (not a code regression) |\n' +
'| 🔧 | The test failed due to mock data issues (not a code regression) |\n' +
'| 🚫 | The test was throttled by API rate limits/overload |\n' +
'| ❌ | The test failed and should be fixed before merging the PR |\n' +
'</details>\n';
}

footer += '\n<details>\n<summary>🔄 <b>Re-run evals manually</b></summary>\n\n' +
'> ⚠️ **Warning:** `/eval` comments always run using the **workflow from master**, not from this PR branch. ' +
'If you modified the GitHub Action (e.g., added secrets or env vars), those changes won\'t take effect.\n>\n' +
'> **To test workflow changes**, use the GitHub CLI or [Actions UI](' + workflowUrl + ') instead:\n>\n' +
'> ```\n> ' + ghCommand + '\n> ```\n\n' +
'---\n\n' +
'**Option 1: Comment on this PR** with `/eval`:\n\n' +
'```\n/eval\nmarkers: regression\n```\n\n' +
'Or with more options (one per line):\n\n' +
Expand All @@ -117,20 +158,17 @@ function buildRerunFooter(p, context) {
'| Option | Description |\n|--------|-------------|\n' +
'| `model` | Model(s) to test (default: same as automatic runs) |\n' +
'| `markers` | Pytest markers (**no default - runs all tests!**) |\n' +
'| `filter` | Pytest -k filter |\n' +
'| `filter` | Pytest -k filter (use `/list` to see valid eval names) |\n' +
'| `iterations` | Number of runs, max 10 |\n' +
'| `branch` | Run evals on a different branch (for cross-branch comparison) |\n\n' +
'**Quick re-run:** Use `/last` to re-run the most recent `/eval` on this PR with the same parameters.\n\n' +
`**Option 2: [Trigger via GitHub Actions UI](${workflowUrl})** → "Run workflow"\n</details>\n` +
'\n<details>\n<summary>🏷️ <b>Valid markers</b></summary>\n\n' +
(p.validMarkers || '_(Collecting from pyproject.toml...)_') +
markersFormatted +
'\n</details>\n' +
'\n<details>\n<summary>📋 <b>Valid eval names (use with filter)</b></summary>\n\n' +
'**test_ask_holmes:**\n' +
(p.askHolmesEvals || '_(Collecting from tests/llm/fixtures/...)_') +
'\n\n**test_investigate:**\n' +
(p.investigateEvals || '_(Collecting from tests/llm/fixtures/...)_') +
'\n</details>\n';
'\n---\n**Commands:** `/eval` · `/last` · `/list`\n';

return footer;
}

module.exports = {
Expand Down
99 changes: 67 additions & 32 deletions .github/workflows/eval-regression.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -36,6 +36,61 @@ permissions:
issues: write

jobs:
# Handle /list command - outputs available eval names
list_evals:
runs-on: ubuntu-latest
if: github.event_name == 'issue_comment' && github.event.issue.pull_request && startsWith(github.event.comment.body, '/list')
steps:
- name: Check permissions
uses: actions/github-script@v7
with:
script: |
const association = context.payload.comment.author_association;
if (!['OWNER', 'MEMBER', 'COLLABORATOR'].includes(association)) {
core.setFailed(`Permission denied: ${association} cannot use /list (requires OWNER, MEMBER, or COLLABORATOR)`);
}

- name: Checkout code
uses: actions/checkout@v4

- name: Post eval list
uses: actions/github-script@v7
with:
script: |
const fs = require('fs');
const path = require('path');

// Collect eval names from fixture directories
const askHolmesPath = 'tests/llm/fixtures/test_ask_holmes';
const investigatePath = 'tests/llm/fixtures/test_investigate';

const getEvalNames = (dir) => {
try {
return fs.readdirSync(dir)
.filter(f => !f.startsWith('.'))
.sort()
.map(f => `\`${f}\``)
.join(', ');
} catch (e) {
return '_(none found)_';
}
};

const askHolmesEvals = getEvalNames(askHolmesPath);
const investigateEvals = getEvalNames(investigatePath);

const body = `## 📋 Available Eval Names\n\n` +
`Use these with \`filter:\` in your \`/eval\` command.\n\n` +
`**test_ask_holmes:**\n${askHolmesEvals}\n\n` +
`**test_investigate:**\n${investigateEvals}`;

await github.rest.issues.createComment({
owner: context.repo.owner,
repo: context.repo.repo,
issue_number: context.issue.number,
body: body
});

llm_evals:
runs-on: ubuntu-latest
if: |
Expand Down Expand Up @@ -335,7 +390,7 @@ jobs:
owner: context.repo.owner,
repo: context.repo.repo,
issue_number: p.prNumber,
body: buildBody(p, progressSteps) + buildRerunFooter(p, context)
body: buildBody(p, progressSteps, { context }) + buildRerunFooter(p, context)
});
core.setOutput('comment_id', comment.data.id.toString());

Expand Down Expand Up @@ -367,7 +422,7 @@ jobs:
owner: context.repo.owner,
repo: context.repo.repo,
comment_id: p.commentId,
body: buildBody(p, progressSteps) + buildRerunFooter(p, context)
body: buildBody(p, progressSteps, { context }) + buildRerunFooter(p, context)
});

- name: Collect evals to run
Expand All @@ -390,22 +445,17 @@ jobs:
TEST_COUNT=$(echo "$TEST_LIST" | grep -c "^tests/llm/" || true)

# Collect valid markers from pyproject.toml using Python/tomllib (excluding 'llm' which is internal)
# Output as comma-separated plain names
VALID_MARKERS=$(python3 << 'PYEOF'
import tomllib
with open('pyproject.toml', 'rb') as f:
config = tomllib.load(f)
markers = config.get('tool', {}).get('pytest', {}).get('ini_options', {}).get('markers', [])
for m in sorted(markers):
name = m.split(':')[0]
if name != 'llm':
print(f'- `{name}`')
names = [m.split(':')[0] for m in sorted(markers) if m.split(':')[0] != 'llm']
print(','.join(names))
PYEOF
)

# Collect valid eval names from fixture directories
ASK_HOLMES_EVALS=$(ls -1 tests/llm/fixtures/test_ask_holmes/ 2>/dev/null | grep -v '^\.' | sort | awk '{print "- \x60" $0 "\x60"}')
INVESTIGATE_EVALS=$(ls -1 tests/llm/fixtures/test_investigate/ 2>/dev/null | grep -v '^\.' | sort | awk '{print "- \x60" $0 "\x60"}')

# Output for next steps
echo "test_count=$TEST_COUNT" >> $GITHUB_OUTPUT

Expand All @@ -419,16 +469,6 @@ jobs:
echo "$VALID_MARKERS" >> $GITHUB_OUTPUT
echo "$EOF" >> $GITHUB_OUTPUT

EOF=$(dd if=/dev/urandom bs=15 count=1 status=none | base64)
echo "ask_holmes_evals<<$EOF" >> $GITHUB_OUTPUT
echo "$ASK_HOLMES_EVALS" >> $GITHUB_OUTPUT
echo "$EOF" >> $GITHUB_OUTPUT

EOF=$(dd if=/dev/urandom bs=15 count=1 status=none | base64)
echo "investigate_evals<<$EOF" >> $GITHUB_OUTPUT
echo "$INVESTIGATE_EVALS" >> $GITHUB_OUTPUT
echo "$EOF" >> $GITHUB_OUTPUT

- name: Update progress - evals collected
if: steps.check-tests.outputs.should-run == 'true' && steps.eval-params.outputs.pr_number != '' && steps.initial-comment.outputs.comment_id != ''
uses: actions/github-script@v7
Expand All @@ -440,9 +480,7 @@ jobs:
comment_id: ${{ toJSON(steps.initial-comment.outputs.comment_id) }},
test_count: ${{ toJSON(steps.test-preview.outputs.test_count) }},
test_preview: ${{ toJSON(steps.test-preview.outputs.test_preview) }},
valid_markers: ${{ toJSON(steps.test-preview.outputs.valid_markers) }},
ask_holmes_evals: ${{ toJSON(steps.test-preview.outputs.ask_holmes_evals) }},
investigate_evals: ${{ toJSON(steps.test-preview.outputs.investigate_evals) }}
valid_markers: ${{ toJSON(steps.test-preview.outputs.valid_markers) }}
});
const progressSteps = [
[true, 'Setup HolmesGPT environment'],
Expand All @@ -454,7 +492,7 @@ jobs:
owner: context.repo.owner,
repo: context.repo.repo,
comment_id: p.commentId,
body: buildBody(p, progressSteps, { testPreview: p.testCount !== '0' ? p.testPreview : null }) + buildRerunFooter(p, context)
body: buildBody(p, progressSteps, { testPreview: p.testCount !== '0' ? p.testPreview : null, context }) + buildRerunFooter(p, context)
});

- name: Setup KIND cluster
Expand All @@ -475,9 +513,7 @@ jobs:
comment_id: ${{ toJSON(steps.initial-comment.outputs.comment_id) }},
test_count: ${{ toJSON(steps.test-preview.outputs.test_count) }},
test_preview: ${{ toJSON(steps.test-preview.outputs.test_preview) }},
valid_markers: ${{ toJSON(steps.test-preview.outputs.valid_markers) }},
ask_holmes_evals: ${{ toJSON(steps.test-preview.outputs.ask_holmes_evals) }},
investigate_evals: ${{ toJSON(steps.test-preview.outputs.investigate_evals) }}
valid_markers: ${{ toJSON(steps.test-preview.outputs.valid_markers) }}
});
const progressSteps = [
[true, 'Setup HolmesGPT environment'],
Expand All @@ -489,7 +525,7 @@ jobs:
owner: context.repo.owner,
repo: context.repo.repo,
comment_id: p.commentId,
body: buildBody(p, progressSteps, { testPreview: p.testCount !== '0' ? p.testPreview : null }) + buildRerunFooter(p, context)
body: buildBody(p, progressSteps, { testPreview: p.testCount !== '0' ? p.testPreview : null, context }) + buildRerunFooter(p, context)
});

- name: Run tests
Expand Down Expand Up @@ -547,8 +583,6 @@ jobs:
comment_id: ${{ toJSON(steps.initial-comment.outputs.comment_id) }},
duration: ${{ toJSON(steps.evals.outputs.duration) }},
valid_markers: ${{ toJSON(steps.test-preview.outputs.valid_markers) }},
ask_holmes_evals: ${{ toJSON(steps.test-preview.outputs.ask_holmes_evals) }},
investigate_evals: ${{ toJSON(steps.test-preview.outputs.investigate_evals) }},
triggered_by: ${{ toJSON(steps.eval-params.outputs.triggered_by) }}
});
const report = fs.existsSync('evals_report.md') ? fs.readFileSync('evals_report.md', 'utf8') : '';
Expand All @@ -564,11 +598,12 @@ jobs:

let body = notifyHeader + buildBody(p, null, {
icon: p.isManual ? '🧪' : '✅',
title: p.isManual ? 'Manual Eval Results' : 'Results of HolmesGPT evals'
title: p.isManual ? 'Manual Eval Results' : 'Results of HolmesGPT evals',
context
});
body += '\n' + (report || '⚠️ No eval report was generated.\n\n');
if (hasFailures) body += `\n### ⚠️ ${failures} Failure${failures === '1' ? '' : 's'} Detected\n\n`;
body += buildRerunFooter(p, context);
body += buildRerunFooter(p, context, { includeLegend: true });

// Delete progress comment and create fresh results comment
// This ensures the user gets a GitHub notification when evals complete
Expand Down
2 changes: 1 addition & 1 deletion tests/llm/utils/reporting/github_reporter.py
Original file line number Diff line number Diff line change
Expand Up @@ -64,7 +64,7 @@ def _generate_historical_details_section(details: HistoricalComparisonDetails) -
Returns:
Markdown string with collapsible details section
"""
lines = ["<details>", "<summary>Historical Comparison Details</summary>\n"]
lines = ["<details>", "<summary><b>Historical Comparison Details</b></summary>\n"]

# Filter description
if details.filter_description:
Expand Down
Loading