Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions .github/dependabot.yml
Original file line number Diff line number Diff line change
Expand Up @@ -8,3 +8,5 @@ updates:
github-actions:
patterns:
- "*"
cooldown:
default-days: 7
33 changes: 27 additions & 6 deletions .github/workflows/benchmark-multinode-tmpl.yml
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,15 @@ name: Template - Multi-Node Benchmark

on:
workflow_call:
secrets:
REPO_PAT:
required: true
INFERENCEX_OFFICIAL_RO_HF_TOKEN:
required: true
MODAL_TOKEN_ID:
required: false
MODAL_TOKEN_SECRET:
required: false
inputs:
config:
description: "Benchmark configuration as JSON"
Expand Down Expand Up @@ -251,7 +260,7 @@ jobs:
- name: Slurm cleanup (pre-run)
run: &slurm-cleanup |
if command -v squeue >/dev/null 2>&1; then
for job_name in "${{ runner.name }}" "inferencex-${{ runner.name }}"; do
for job_name in "${RUNNER_NAME}" "inferencex-${RUNNER_NAME}"; do
echo "[Slurm] Cleaning up jobs named: $job_name ..."
scancel --user="$USER" --name="$job_name" || true
while [ -n "$(squeue --user="$USER" --name="$job_name" --noheader --format='%i')" ]; do
Expand All @@ -272,7 +281,7 @@ jobs:
# module git dir — a truncated clone may lack HEAD entirely, and a
# git dir whose HEAD does not resolve to a commit is dropped
# together with its worktree so checkout re-clones it.
repo_dir="${{ github.workspace }}"
repo_dir="${GITHUB_WORKSPACE}"
if [ -d "${repo_dir}/.git" ]; then
find "${repo_dir}/.git" -name index.lock -type f -delete || true
for gitdir in "${repo_dir}"/.git/modules/* "${repo_dir}"/.git/modules/*/*; do
Expand All @@ -293,9 +302,12 @@ jobs:
ref: ${{ inputs.ref || github.sha }}
clean: true
submodules: true
persist-credentials: false

- name: Launch multi-node job script
env:
PREFILL_ADDITIONAL_SETTINGS: ${{ toJSON(fromJSON(inputs.config).prefill.additional-settings) }}
DECODE_ADDITIONAL_SETTINGS: ${{ toJSON(fromJSON(inputs.config).decode.additional-settings) }}
RUNNER_NAME: ${{ runner.name }}
RUNNER_TYPE: ${{ inputs.runner }}
RESULT_FILENAME_BASE: ${{ env.EXP_NAME }}_${{ env.PRECISION }}_${{ env.FRAMEWORK }}_prefill-tp${{ env.PREFILL_TP }}-pp${{ env.PREFILL_PP_SIZE }}-dcp${{ env.PREFILL_DCP_SIZE }}-pcp${{ env.PREFILL_PCP_SIZE }}-ep${{ env.PREFILL_EP }}-dp${{ env.PREFILL_DP_ATTN }}-nw${{ env.PREFILL_NUM_WORKERS }}_decode-tp${{ env.DECODE_TP }}-pp${{ env.DECODE_PP_SIZE }}-dcp${{ env.DECODE_DCP_SIZE }}-pcp${{ env.DECODE_PCP_SIZE }}-ep${{ env.DECODE_EP }}-dp${{ env.DECODE_DP_ATTN }}-nw${{ env.DECODE_NUM_WORKERS }}_disagg-${{ env.DISAGG }}_spec-${{ env.SPEC_DECODING }}_conc${{ join(fromJson(inputs.conc-list), 'x') }}_${{ runner.name }}
Expand Down Expand Up @@ -326,22 +338,31 @@ jobs:
if [[ -f benchmarks/multi_node/runtime_settings.sh ]]; then
source benchmarks/multi_node/runtime_settings.sh
fi
export ${{ join(fromJSON(inputs.config).prefill.additional-settings, ' ') }} ${{ join(fromJSON(inputs.config).decode.additional-settings, ' ') }}
# Assign data literally; recipe values must never become shell code.
settings_json=$(jq -cen \
--argjson prefill "$PREFILL_ADDITIONAL_SETTINGS" \
--argjson decode "$DECODE_ADDITIONAL_SETTINGS" \
'($prefill // []) + ($decode // []) |
if all(.[]; type == "string" and test("^[A-Za-z_][A-Za-z0-9_]*=")) then .
else error("additional-settings must contain NAME=value assignments") end')
while IFS= read -r -d '' setting; do
export "$setting"
done < <(jq -j '.[] + "\u0000"' <<< "$settings_json")
# Resolve the workflow's documented automatic eval-concurrency selection.
if [[ -z "$EVAL_CONC" ]]; then
EVAL_CONC=$(python3 -c 'import os; print(max(map(int, os.environ["CONC_LIST"].split())))')
fi
export EVAL_CONC
export IS_MULTINODE=true
bash ./runners/launch_${RUNNER_NAME%%_*}.sh
if [ "${{ inputs.eval-only }}" = "true" ]; then
bash "./runners/launch_${RUNNER_NAME%%_*}.sh"
if [ "${EVAL_ONLY}" = "true" ]; then
echo "Eval-only mode: skipping benchmark result file check"
# Verify eval produced results
if ! ls results*.json 1>/dev/null 2>&1; then
echo "Eval-only run failed: no results*.json files found." >&2
exit 1
fi
elif [ "${{ inputs.scenario-type }}" = "agentic-coding" ]; then
elif [ "${SCENARIO_TYPE}" = "agentic-coding" ]; then
expected_count=$(wc -w <<< "$CONC_LIST" | tr -d ' ')
shopt -s nullglob
agentic_results=("${RESULT_FILENAME}"_conc*.json)
Expand Down
49 changes: 33 additions & 16 deletions .github/workflows/benchmark-tmpl.yml
Original file line number Diff line number Diff line change
@@ -1,6 +1,15 @@
name: Template - Benchmark
on:
workflow_call:
secrets:
REPO_PAT:
required: true
INFERENCEX_OFFICIAL_RO_HF_TOKEN:
required: true
MODAL_TOKEN_ID:
required: false
MODAL_TOKEN_SECRET:
required: false
inputs:
config:
description: "Benchmark configuration as JSON"
Expand Down Expand Up @@ -163,15 +172,15 @@ jobs:
format(
'["self-hosted",{0},{1},{2},{3},{4}]',
toJSON(inputs.runner),
toJSON('nodes:1'),
'"nodes:1"',
toJSON(format('ci-job-{0}-{1}', inputs.priority, inputs.queue-token)),
toJSON(format('ci-attempt-{0}', github.run_attempt)),
toJSON(format('ci-skip-queue-pr-{0}', inputs.skip-queue-pr))
) ||
format(
'["self-hosted",{0},{1},{2},{3}]',
toJSON(inputs.runner),
toJSON('nodes:1'),
'"nodes:1"',
toJSON(format('ci-job-{0}-{1}', inputs.priority, inputs.queue-token)),
toJSON(format('ci-attempt-{0}', github.run_attempt))
)
Expand Down Expand Up @@ -204,15 +213,15 @@ jobs:
${{ inputs.scenario-type == 'agentic-coding' && fromJSON(inputs.config).kv-offload-backend.name != 'none' && fromJSON(inputs.config).kv-offload-backend.name != 'default' && format('{0}', fromJSON(inputs.config).kv-offload-backend.name) || '' }}
c${{ fromJSON(inputs.config).conc }}${{ inputs.eval-only && ' | eval-only' || (inputs.run-eval && ' | eval' || '') }}
steps:
- name: Resource cleanup (pre-run)
run: &resource-cleanup |
# Containerized AMD Slurm jobs can leave root-owned artifacts when
# interrupted before their launcher EXIT trap runs. Repair ownership
# before checkout cleanup and again after the job via this shared step.
if [[ "${{ inputs.runner }}" == "cluster:mi355x-amds" && -d "$GITHUB_WORKSPACE" ]]; then
- name: Repair root-owned artifacts (pre-run)
if: ${{ inputs.runner == 'cluster:mi355x-amds' }}
run: |
if [[ -d "$GITHUB_WORKSPACE" ]]; then
sudo chown -R "$(id -u):$(id -g)" "$GITHUB_WORKSPACE"
fi

- name: Resource cleanup (pre-run)
run: &resource-cleanup |
# Cleanup Docker resources
if command -v docker >/dev/null 2>&1 && docker info >/dev/null 2>&1; then
echo "[Docker] Cleaning up resources ..."
Expand All @@ -226,10 +235,10 @@ jobs:

# Cleanup SLURM resources
if command -v squeue >/dev/null 2>&1; then
echo "[Slurm] Cleaning up jobs with name: ${{ runner.name }} ..."
scancel --name="${{ runner.name }}" || true
while [ -n "$(squeue --name='${{ runner.name }}' --noheader --format='%i')" ]; do
squeue --name="${{ runner.name }}"
echo "[Slurm] Cleaning up jobs with name: ${RUNNER_NAME} ..."
scancel --name="${RUNNER_NAME}" || true
while [ -n "$(squeue --name="${RUNNER_NAME}" --noheader --format='%i')" ]; do
squeue --name="${RUNNER_NAME}"
sleep 5
done
fi
Expand All @@ -245,7 +254,7 @@ jobs:
# module git dir — a truncated clone may lack HEAD entirely, and a
# git dir whose HEAD does not resolve to a commit is dropped
# together with its worktree so checkout re-clones it.
repo_dir="${{ github.workspace }}"
repo_dir="${GITHUB_WORKSPACE}"
if [ -d "${repo_dir}/.git" ]; then
find "${repo_dir}/.git" -name index.lock -type f -delete || true
for gitdir in "${repo_dir}"/.git/modules/* "${repo_dir}"/.git/modules/*/*; do
Expand All @@ -266,6 +275,7 @@ jobs:
ref: ${{ inputs.ref || github.sha }}
clean: true
submodules: true
persist-credentials: false

- name: Launch job script
env:
Expand Down Expand Up @@ -297,9 +307,9 @@ jobs:
if [[ -f runners/runtime_settings.sh ]]; then
source runners/runtime_settings.sh
fi
bash ./runners/launch_${RUNNER_NAME%%_*}.sh
bash "./runners/launch_${RUNNER_NAME%%_*}.sh"

if [ "${{ inputs.eval-only }}" = "true" ]; then
if [ "${EVAL_ONLY}" = "true" ]; then
echo "Eval-only mode: skipping benchmark result file check"
# Verify eval produced results
if ! ls results*.json 1>/dev/null 2>&1; then
Expand All @@ -322,7 +332,7 @@ jobs:
exit 1
fi

if [ "${{ inputs.scenario-type }}" = "agentic-coding" ]; then
if [ "${SCENARIO_TYPE}" = "agentic-coding" ]; then
python3 -m utils.agentic.validation.validate_agentic_result \
results/aiperf_artifacts \
--failed-request-threshold "$AIPERF_FAILED_REQUEST_THRESHOLD"
Expand Down Expand Up @@ -451,6 +461,13 @@ jobs:
rm -f -- ./*_artifacts.tar.gz || true
rm -f agent_preds.json predictions.jsonl swebench_report_*.json *.traj* || true

- name: Repair root-owned artifacts (post-run)
if: ${{ always() && inputs.runner == 'cluster:mi355x-amds' }}
run: |
if [[ -d "$GITHUB_WORKSPACE" ]]; then
sudo chown -R "$(id -u):$(id -g)" "$GITHUB_WORKSPACE"
fi

- name: Resource cleanup (post-run)
if: always()
run: *resource-cleanup
9 changes: 7 additions & 2 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -3,10 +3,15 @@ name: CI
on:
pull_request:
types: [opened, synchronize, reopened, ready_for_review]
paths: ['**/*.py']
paths: &python-paths
- '**/*.py'
- '.github/workflows/ci.yml'
- '.claude/requirements-mcp.txt'
- 'infx/ruff.toml'
- '**/pytest.ini'
push:
branches: [main]
paths: ['**/*.py']
paths: *python-paths
workflow_dispatch:

permissions:
Expand Down
Loading