diff --git a/.github/dependabot.yml b/.github/dependabot.yml index 079bd6d278..a626b834cd 100644 --- a/.github/dependabot.yml +++ b/.github/dependabot.yml @@ -8,3 +8,5 @@ updates: github-actions: patterns: - "*" + cooldown: + default-days: 7 diff --git a/.github/workflows/benchmark-multinode-tmpl.yml b/.github/workflows/benchmark-multinode-tmpl.yml index 81f8233377..3978698f98 100644 --- a/.github/workflows/benchmark-multinode-tmpl.yml +++ b/.github/workflows/benchmark-multinode-tmpl.yml @@ -2,6 +2,15 @@ name: Template - Multi-Node Benchmark on: workflow_call: + secrets: + REPO_PAT: + required: true + INFERENCEX_OFFICIAL_RO_HF_TOKEN: + required: true + MODAL_TOKEN_ID: + required: false + MODAL_TOKEN_SECRET: + required: false inputs: config: description: "Benchmark configuration as JSON" @@ -251,7 +260,7 @@ jobs: - name: Slurm cleanup (pre-run) run: &slurm-cleanup | if command -v squeue >/dev/null 2>&1; then - for job_name in "${{ runner.name }}" "inferencex-${{ runner.name }}"; do + for job_name in "${RUNNER_NAME}" "inferencex-${RUNNER_NAME}"; do echo "[Slurm] Cleaning up jobs named: $job_name ..." scancel --user="$USER" --name="$job_name" || true while [ -n "$(squeue --user="$USER" --name="$job_name" --noheader --format='%i')" ]; do @@ -272,7 +281,7 @@ jobs: # module git dir — a truncated clone may lack HEAD entirely, and a # git dir whose HEAD does not resolve to a commit is dropped # together with its worktree so checkout re-clones it. - repo_dir="${{ github.workspace }}" + repo_dir="${GITHUB_WORKSPACE}" if [ -d "${repo_dir}/.git" ]; then find "${repo_dir}/.git" -name index.lock -type f -delete || true for gitdir in "${repo_dir}"/.git/modules/* "${repo_dir}"/.git/modules/*/*; do @@ -293,9 +302,12 @@ jobs: ref: ${{ inputs.ref || github.sha }} clean: true submodules: true + persist-credentials: false - name: Launch multi-node job script env: + PREFILL_ADDITIONAL_SETTINGS: ${{ toJSON(fromJSON(inputs.config).prefill.additional-settings) }} + DECODE_ADDITIONAL_SETTINGS: ${{ toJSON(fromJSON(inputs.config).decode.additional-settings) }} RUNNER_NAME: ${{ runner.name }} RUNNER_TYPE: ${{ inputs.runner }} RESULT_FILENAME_BASE: ${{ env.EXP_NAME }}_${{ env.PRECISION }}_${{ env.FRAMEWORK }}_prefill-tp${{ env.PREFILL_TP }}-pp${{ env.PREFILL_PP_SIZE }}-dcp${{ env.PREFILL_DCP_SIZE }}-pcp${{ env.PREFILL_PCP_SIZE }}-ep${{ env.PREFILL_EP }}-dp${{ env.PREFILL_DP_ATTN }}-nw${{ env.PREFILL_NUM_WORKERS }}_decode-tp${{ env.DECODE_TP }}-pp${{ env.DECODE_PP_SIZE }}-dcp${{ env.DECODE_DCP_SIZE }}-pcp${{ env.DECODE_PCP_SIZE }}-ep${{ env.DECODE_EP }}-dp${{ env.DECODE_DP_ATTN }}-nw${{ env.DECODE_NUM_WORKERS }}_disagg-${{ env.DISAGG }}_spec-${{ env.SPEC_DECODING }}_conc${{ join(fromJson(inputs.conc-list), 'x') }}_${{ runner.name }} @@ -326,22 +338,31 @@ jobs: if [[ -f benchmarks/multi_node/runtime_settings.sh ]]; then source benchmarks/multi_node/runtime_settings.sh fi - export ${{ join(fromJSON(inputs.config).prefill.additional-settings, ' ') }} ${{ join(fromJSON(inputs.config).decode.additional-settings, ' ') }} + # Assign data literally; recipe values must never become shell code. + settings_json=$(jq -cen \ + --argjson prefill "$PREFILL_ADDITIONAL_SETTINGS" \ + --argjson decode "$DECODE_ADDITIONAL_SETTINGS" \ + '($prefill // []) + ($decode // []) | + if all(.[]; type == "string" and test("^[A-Za-z_][A-Za-z0-9_]*=")) then . + else error("additional-settings must contain NAME=value assignments") end') + while IFS= read -r -d '' setting; do + export "$setting" + done < <(jq -j '.[] + "\u0000"' <<< "$settings_json") # Resolve the workflow's documented automatic eval-concurrency selection. if [[ -z "$EVAL_CONC" ]]; then EVAL_CONC=$(python3 -c 'import os; print(max(map(int, os.environ["CONC_LIST"].split())))') fi export EVAL_CONC export IS_MULTINODE=true - bash ./runners/launch_${RUNNER_NAME%%_*}.sh - if [ "${{ inputs.eval-only }}" = "true" ]; then + bash "./runners/launch_${RUNNER_NAME%%_*}.sh" + if [ "${EVAL_ONLY}" = "true" ]; then echo "Eval-only mode: skipping benchmark result file check" # Verify eval produced results if ! ls results*.json 1>/dev/null 2>&1; then echo "Eval-only run failed: no results*.json files found." >&2 exit 1 fi - elif [ "${{ inputs.scenario-type }}" = "agentic-coding" ]; then + elif [ "${SCENARIO_TYPE}" = "agentic-coding" ]; then expected_count=$(wc -w <<< "$CONC_LIST" | tr -d ' ') shopt -s nullglob agentic_results=("${RESULT_FILENAME}"_conc*.json) diff --git a/.github/workflows/benchmark-tmpl.yml b/.github/workflows/benchmark-tmpl.yml index c19bd99c86..baa0efcfbc 100644 --- a/.github/workflows/benchmark-tmpl.yml +++ b/.github/workflows/benchmark-tmpl.yml @@ -1,6 +1,15 @@ name: Template - Benchmark on: workflow_call: + secrets: + REPO_PAT: + required: true + INFERENCEX_OFFICIAL_RO_HF_TOKEN: + required: true + MODAL_TOKEN_ID: + required: false + MODAL_TOKEN_SECRET: + required: false inputs: config: description: "Benchmark configuration as JSON" @@ -163,7 +172,7 @@ jobs: format( '["self-hosted",{0},{1},{2},{3},{4}]', toJSON(inputs.runner), - toJSON('nodes:1'), + '"nodes:1"', toJSON(format('ci-job-{0}-{1}', inputs.priority, inputs.queue-token)), toJSON(format('ci-attempt-{0}', github.run_attempt)), toJSON(format('ci-skip-queue-pr-{0}', inputs.skip-queue-pr)) @@ -171,7 +180,7 @@ jobs: format( '["self-hosted",{0},{1},{2},{3}]', toJSON(inputs.runner), - toJSON('nodes:1'), + '"nodes:1"', toJSON(format('ci-job-{0}-{1}', inputs.priority, inputs.queue-token)), toJSON(format('ci-attempt-{0}', github.run_attempt)) ) @@ -204,15 +213,15 @@ jobs: ${{ inputs.scenario-type == 'agentic-coding' && fromJSON(inputs.config).kv-offload-backend.name != 'none' && fromJSON(inputs.config).kv-offload-backend.name != 'default' && format('{0}', fromJSON(inputs.config).kv-offload-backend.name) || '' }} c${{ fromJSON(inputs.config).conc }}${{ inputs.eval-only && ' | eval-only' || (inputs.run-eval && ' | eval' || '') }} steps: - - name: Resource cleanup (pre-run) - run: &resource-cleanup | - # Containerized AMD Slurm jobs can leave root-owned artifacts when - # interrupted before their launcher EXIT trap runs. Repair ownership - # before checkout cleanup and again after the job via this shared step. - if [[ "${{ inputs.runner }}" == "cluster:mi355x-amds" && -d "$GITHUB_WORKSPACE" ]]; then + - name: Repair root-owned artifacts (pre-run) + if: ${{ inputs.runner == 'cluster:mi355x-amds' }} + run: | + if [[ -d "$GITHUB_WORKSPACE" ]]; then sudo chown -R "$(id -u):$(id -g)" "$GITHUB_WORKSPACE" fi + - name: Resource cleanup (pre-run) + run: &resource-cleanup | # Cleanup Docker resources if command -v docker >/dev/null 2>&1 && docker info >/dev/null 2>&1; then echo "[Docker] Cleaning up resources ..." @@ -226,10 +235,10 @@ jobs: # Cleanup SLURM resources if command -v squeue >/dev/null 2>&1; then - echo "[Slurm] Cleaning up jobs with name: ${{ runner.name }} ..." - scancel --name="${{ runner.name }}" || true - while [ -n "$(squeue --name='${{ runner.name }}' --noheader --format='%i')" ]; do - squeue --name="${{ runner.name }}" + echo "[Slurm] Cleaning up jobs with name: ${RUNNER_NAME} ..." + scancel --name="${RUNNER_NAME}" || true + while [ -n "$(squeue --name="${RUNNER_NAME}" --noheader --format='%i')" ]; do + squeue --name="${RUNNER_NAME}" sleep 5 done fi @@ -245,7 +254,7 @@ jobs: # module git dir — a truncated clone may lack HEAD entirely, and a # git dir whose HEAD does not resolve to a commit is dropped # together with its worktree so checkout re-clones it. - repo_dir="${{ github.workspace }}" + repo_dir="${GITHUB_WORKSPACE}" if [ -d "${repo_dir}/.git" ]; then find "${repo_dir}/.git" -name index.lock -type f -delete || true for gitdir in "${repo_dir}"/.git/modules/* "${repo_dir}"/.git/modules/*/*; do @@ -266,6 +275,7 @@ jobs: ref: ${{ inputs.ref || github.sha }} clean: true submodules: true + persist-credentials: false - name: Launch job script env: @@ -297,9 +307,9 @@ jobs: if [[ -f runners/runtime_settings.sh ]]; then source runners/runtime_settings.sh fi - bash ./runners/launch_${RUNNER_NAME%%_*}.sh + bash "./runners/launch_${RUNNER_NAME%%_*}.sh" - if [ "${{ inputs.eval-only }}" = "true" ]; then + if [ "${EVAL_ONLY}" = "true" ]; then echo "Eval-only mode: skipping benchmark result file check" # Verify eval produced results if ! ls results*.json 1>/dev/null 2>&1; then @@ -322,7 +332,7 @@ jobs: exit 1 fi - if [ "${{ inputs.scenario-type }}" = "agentic-coding" ]; then + if [ "${SCENARIO_TYPE}" = "agentic-coding" ]; then python3 -m utils.agentic.validation.validate_agentic_result \ results/aiperf_artifacts \ --failed-request-threshold "$AIPERF_FAILED_REQUEST_THRESHOLD" @@ -451,6 +461,13 @@ jobs: rm -f -- ./*_artifacts.tar.gz || true rm -f agent_preds.json predictions.jsonl swebench_report_*.json *.traj* || true + - name: Repair root-owned artifacts (post-run) + if: ${{ always() && inputs.runner == 'cluster:mi355x-amds' }} + run: | + if [[ -d "$GITHUB_WORKSPACE" ]]; then + sudo chown -R "$(id -u):$(id -g)" "$GITHUB_WORKSPACE" + fi + - name: Resource cleanup (post-run) if: always() run: *resource-cleanup diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 1c76e4509e..67187a6022 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -3,10 +3,15 @@ name: CI on: pull_request: types: [opened, synchronize, reopened, ready_for_review] - paths: ['**/*.py'] + paths: &python-paths + - '**/*.py' + - '.github/workflows/ci.yml' + - '.claude/requirements-mcp.txt' + - 'infx/ruff.toml' + - '**/pytest.ini' push: branches: [main] - paths: ['**/*.py'] + paths: *python-paths workflow_dispatch: permissions: diff --git a/.github/workflows/claude-pr-review.yml b/.github/workflows/claude-pr-review.yml deleted file mode 100644 index 1573c62ba3..0000000000 --- a/.github/workflows/claude-pr-review.yml +++ /dev/null @@ -1,349 +0,0 @@ -name: PR Review - -on: - pull_request: - types: [ready_for_review] - issue_comment: - types: [created] - pull_request_review_comment: - types: [created] - -concurrency: - group: pr-review-${{ github.event.pull_request.number }} - cancel-in-progress: false - -jobs: - review: - runs-on: ubuntu-latest - # Only run if: - # 1. It's a PR event from someone with write access, OR - # 2. It's a comment containing @pr-claude from someone with write access - if: | - (github.event_name == 'pull_request' && - contains(fromJson('["OWNER", "MEMBER", "COLLABORATOR"]'), github.event.pull_request.author_association)) || - ((github.event_name == 'issue_comment' || github.event_name == 'pull_request_review_comment') && - contains(github.event.comment.body, '@pr-claude') && - contains(fromJson('["OWNER", "MEMBER", "COLLABORATOR"]'), github.event.comment.author_association)) - permissions: - contents: read - pull-requests: write - actions: read - id-token: write - steps: - - name: Checkout repository - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - with: - fetch-depth: 0 - - - name: Set up uv - uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1 - - - name: Setup Node.js - uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0 - with: - node-version: '22' - package-manager-cache: false - - - name: Install Claude Code - id: claude-code - run: | - # The native installer can report success without creating its launcher. - npm install --prefix "$RUNNER_TEMP/claude-code" --no-audit --no-fund @anthropic-ai/claude-code@2.1.265 - claude_path="$RUNNER_TEMP/claude-code/node_modules/.bin/claude" - "$claude_path" --version - echo "path=$claude_path" >> "$GITHUB_OUTPUT" - - - name: PR Review with Claude - uses: anthropics/claude-code-action@0d0e0876d3eaa933f45dc692f7a4312c83caf36f # v1.0.218 - env: - INFERENCEMAX_ROOT: ${{ github.workspace }} - with: - anthropic_api_key: ${{ secrets.ANTHROPIC_API_KEY }} - path_to_claude_code_executable: ${{ steps.claude-code.outputs.path }} - trigger_phrase: "@pr-claude" - track_progress: true - allowed_bots: '' - settings: | - {"fastMode": true} - - claude_args: | - --model 'claude-fable-5-1' - --mcp-config .github/mcp-ci.json - --allowedTools "mcp__github_inline_comment__create_inline_comment,mcp__inferencemax-repos__*,Bash(gh pr comment:*),Bash(gh pr diff:*),Bash(gh pr view:*)" - prompt: | - REPO: ${{ github.repository }} - PR NUMBER: ${{ github.event.pull_request.number || github.event.issue.number }} - - You are reviewing code for InferenceMAX. Your job is to provide HIGH-SIGNAL feedback only. - - ## Commands: - - `@pr-claude review` - Full review of the PR - - `@pr-claude re-review` - Re-review only NEW changes since your last review (check your previous comments first) - - `@pr-claude review ` - Review only a specific file - - `@pr-claude` followed by a question - Answer the question about this PR - - ## If this is a re-review: - 1. First, check existing review comments on this PR using `gh pr view` - 2. Focus ONLY on new commits or changes not previously reviewed - 3. Do NOT repeat previous feedback - reference it if still applicable - 4. If previous issues were fixed, acknowledge briefly in the summary - - ## ONLY comment when you find: - 1. **Bugs**: Code that is broken, will crash, or produces incorrect results - 2. **Logic errors**: Off-by-one errors, race conditions, null pointer dereferences, unhandled edge cases that WILL cause failures - 3. **Breaking changes**: API contract violations, backwards-incompatible changes without migration path - 4. **Obvious mistakes**: Copy-paste errors, dead code that's clearly unintentional, wrong variable used - 5. **Resource leaks**: Unclosed connections, missing cleanup, memory leaks - - ## DO NOT comment on: - - Style preferences or formatting (we have linters for that) - - "Consider doing X" suggestions unless the current code is actually broken - - Minor naming nitpicks - - Adding more comments or documentation - - Theoretical performance improvements without evidence of actual impact - - "Best practices" that don't apply to this specific context - - Praise or positive feedback (save it for the summary) - - Issues you already commented on in a previous review - - ## Comment format: - For each issue, use inline comments with this format: - **[SEVERITY]**: Brief description of the actual problem - **Why it matters**: What will break or go wrong - **Fix**: Concrete suggestion (not vague advice) - **Fix** When possible, the fix should use the GitHub Multi line Code Suggestion. Here is an example on how to use it - - start with 3 backtick symbols followed by the keyword suggestion and then for lines to suggest to delete use the minus symbol and for lines to suggest add, use the plus symbol - - ```suggestion - - line to delete - + line to add - ``` - - Severity levels: - - 🔴 **BLOCKING**: Must fix before merge - will cause bugs/crashes/security issues - - 🟡 **WARNING**: Should fix - likely to cause problems in edge cases - - ## Output: - - Use `mcp__github_inline_comment__create_inline_comment` for specific code issues - - Use `gh pr comment` ONCE at the end for a brief summary (max 3-4 sentences) - - If the PR looks good with no issues, just say "LGTM - no blocking issues found" and nothing else - - For re-reviews, prefix summary with "Re-review:" and note what changed - - ## Master Config and Perf Changelog Validation: - When reviewing a PR, check if any of the following master config files were modified: - - `configs/amd-master.yaml` - - `configs/nvidia-master.yaml` - - If either master config file was edited AND `perf-changelog.yaml` was NOT edited in the same PR: - - This is a 🔴 **BLOCKING** issue - - Comment that `perf-changelog.yaml` must also be updated when master config files are changed - - The perf-changelog entry should document what changed in the config and include the PR link - - Format: "Master config files were modified but `perf-changelog.yaml` was not updated. When changing `configs/amd-master.yaml` or `configs/nvidia-master.yaml`, you must add a corresponding entry to `perf-changelog.yaml` documenting the changes." - - ### Perf Changelog Entry Position: - `perf-changelog.yaml` is read in chronological order — oldest entries at the top, newest at the bottom. New entries MUST be appended to the END of the file. Never insert in the middle or prepend. - - When reviewing a PR diff for `perf-changelog.yaml`: - - Check that every newly added entry lands at the very end of the file (i.e., the diff hunk adds lines after the previous last entry, not before existing entries). - - If any new entry is inserted above/between existing entries rather than appended to the end: - - This is a 🔴 **BLOCKING** issue - - Comment: "New `perf-changelog.yaml` entries must be appended to the END of the file. The file is read chronologically (oldest at top, newest at bottom), so inserting in the middle or prepending breaks the ordering. Please move the new entry(ies) to the bottom of the file." - - ### Append-only Perf Changelog Safety: - When a new `perf-changelog.yaml` entry contains `append-only: true`, verify the complete PR diff before approving it: - - Do not use a file allowlist. Supporting code, benchmark scripts, launchers, helpers, and other files may change. Inspect the complete diff and judge whether each benchmark-affecting change is behaviorally isolated to the appended points. - - Every newly added changelog entry must contain `append-only: true`; append-only and regular entries may not be mixed. - - Treat the generated base matrix as an immutable subset of the generated head matrix. Every existing point must remain present with the same image and complete recipe. Additions may include new concurrency values or entirely new recipe variants (for example, a new tensor-parallelism value) inside the selected existing config/scenario, but they must retain the existing visual curve's single non-null image. - - Benchmark or launch logic may change only when every changed behavior is on a control-flow path uniquely gated to the corresponding newly appended points. Trace the selected config and generated runtime values through every affected file into the condition. Confirm the path cannot be reached by any existing point. Unguarded/shared setup changes, or a branch also used by an existing concurrency/config/scenario, are blocking. - - No existing point, recipe variant, config, or scenario may be removed or replaced. New configs and scenarios are out of scope for append-only mode; new generated variants inside the selected existing config/scenario are allowed. - - Eval modifiers (`evals-only`, `all-evals`, `eval-min-prefill-ep`) are not allowed. - If any condition fails, report a 🔴 **BLOCKING** issue. Never reject a change merely because of its file path; reject it when its benchmark effect is not exclusive to the appended points or the exclusivity cannot be proven from the diff. - - ## Terminology: - - **STP (Single Token Prediction)**: Standard autoregressive decoding — one token per forward pass. No speculative decoding or MTP. Benchmarks labeled "STP only" use vanilla decoding. - - **MTP (Multi-Token Prediction)**: Predicts multiple tokens per forward pass using speculative decoding (e.g., EAGLE, NEXTN). - - ## Expert Parallelism Validation: - When reviewing benchmark scripts for MoE models, verify that expert parallelism flags are used correctly: - - - **vLLM** (`vllm serve`): Uses `--enable-expert-parallel` (boolean flag). Does NOT accept `--expert-parallel-size`. EP size is automatically determined by vLLM. - - **SGLang** (`sglang.launch_server`): Uses `--expert-parallel-size N` (explicit integer). - - **ATOM** (AMD vLLM fork): Uses `--enable-expert-parallel` (same as vLLM). - - **Required pattern for vLLM/ATOM scripts:** - Scripts must NOT hardcode `--enable-expert-parallel`. Instead, they should conditionally enable it based on the `EP_SIZE` env var: - ```bash - if [ "$EP_SIZE" -gt 1 ]; then - EP=" --enable-expert-parallel" - else - EP=" " - fi - ``` - If a script hardcodes `--enable-expert-parallel` without checking `EP_SIZE`: - - This is a 🟡 **WARNING** issue - - Comment: "Expert parallelism should be conditional on the `EP_SIZE` env var from the config YAML, not hardcoded. Use the `if [ \"$EP_SIZE\" -gt 1 ]` pattern to conditionally enable `--enable-expert-parallel`." - - Remember: Silence is golden. No comment is better than a low-value comment. - - ## Container Image Accessibility Validation: - When reviewing changes to `configs/*-master.yaml` files, verify that ALL `image:` values are publicly accessible: - - **Valid image formats (publicly accessible):** - - Docker Hub: `organization/image:tag` (e.g., `lmsysorg/sglang:v0.5.7-rocm700-mi35x`) - - NGC: `nvcr.io/nvidia/...` or `nvcr.io#nvidia/...` (e.g., `nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post1` or `nvcr.io#nvidia/tensorrt-llm/release:1.1.0rc2.post2`) - - Other public registries: `ghcr.io/...`, `quay.io/...`, `rocm/...` - - **Invalid image formats (NOT publicly accessible):** - - generally these images are not best practices to have in: - - Local file paths: `/scratch/...`, `/home/...`, `/data/...`, or any path starting with `/` - - `.sqsh` files (squashfs containers stored locally) - - Internal/private registry paths that are not publicly resolvable - - If any `image:` field contains a local path or non-public image: - - This is a 🔴 **BLOCKING** issue - - Comment: "Image must be publicly accessible on NGC, Docker Hub, or another public registry. Local paths like `/scratch/...` or `.sqsh` files are generally not accepted. Please push the container to a public registry (e.g., `nvcr.io/nvidia/...` for NGC) and update the config with the public image reference." - - Link to the specific line with the invalid image path - - ## Enroot Import Validation for Launch Scripts: - When reviewing changes to `runners/launch_*.sh` files, verify that the script properly transforms public Docker images to enroot local images for reproducibility. - - **Expected pattern:** - The script should include an enroot import command like: - ```bash - srun --jobid=$JOB_ID bash -c "enroot import -o $SQUASH_FILE docker://$IMAGE" - ``` - or similar variations such as: - ```bash - srun -N 1 -A $SLURM_ACCOUNT -p $SLURM_PARTITION bash -c "enroot import -o $SQUASH_FILE docker://$IMAGE" - ``` - - **Why this matters:** - - Ensures the exact same public NGC/Docker Hub image is used - - Makes benchmarks reproducible by anyone with access to the public image - - Prevents reliance on pre-existing local container images that others cannot access - - **Validation Steps:** - 1. Look for `enroot import` commands that convert `docker://` images to local `.sqsh` files - 2. The image source should be a public registry (NGC, Docker Hub, etc.), not a local path - 3. If the script uses containers but does NOT have an `enroot import docker://` pattern: - - This is a 🟡 **WARNING** issue - - Comment: "This launch script uses container images but does not appear to transform a public Docker image to an enroot local image using `enroot import -o $SQUASH_FILE docker://$IMAGE`. For reproducibility, please either: - 1. Add the enroot import pattern to pull from a public registry, OR - 2. Explain why this script has a different workflow (e.g., uses a different container runtime, pre-built images are acceptable for this use case, etc.)" - - Ask the developer to provide a reasonable explanation if the pattern is intentionally omitted - - ## vLLM and SGLang Source Code Access: - You have access to vLLM and SGLang source code via the inferencemax-repos MCP server: - - Use `mcp__inferencemax-repos__*` tools to access repository source code - - Resources are available via URIs: `vllm:///path/to/file.py` and `sglang:///path/to/file.py` - - The server automatically detects and checks out the version matching InferenceMAX configs - - Use the `list_versions` tool to see detected versions - - Use the `switch_version` tool to switch to a different version if needed - - **When to use this:** - - When reviewing changes that interact with vLLM/SGLang APIs or internals - - To verify that code correctly uses vLLM/SGLang features - - To check if assumptions about vLLM/SGLang behavior are accurate - - To understand how vLLM/SGLang implements features being used/modified - - To cross-reference documentation claims with actual implementation - - **Example use cases:** - - PR modifies vLLM scheduler usage → Check actual vLLM scheduler implementation - - PR adds SGLang integration → Verify SGLang API matches the PR's usage - - PR claims feature X works a certain way → Read the source to confirm - - PR has a bug related to KV cache → Look at vLLM's KV cache implementation - - ## Line Count Report for the matrix generator: - If `infx/matrix/generate.py` was modified in this PR: - 1. Count the total lines in the current (PR) version of the file - 2. Count the lines in the base branch version of `infx/matrix/generate.py`. If that path does not exist in the base revision, use `utils/matrix_logic/generate_sweep_configs.py` from that revision instead. This fallback compares the implementation across the package migration rather than counting the new path as entirely new code. - 3. Calculate the difference (current - base) - 4. Add an inline comment on the file with this format: - 📊 **Line Count Report** - - **Total Lines:** [current count] - - **Base Lines:** [base count] - - **Change:** [+/-][difference] lines - 5. Use 📈 for additions, 📉 for removals, ➡️ for no change - - ## Benchmark Script Code Style Guidelines: - When reviewing changes to `benchmarks/*.sh` files, verify the following code style requirements: - - ### 1. Server Launch Command Formatting: - All `sglang.launch_server` and `vllm serve` commands MUST have their arguments formatted on separate lines for readability. - - **Invalid format (all arguments on one line):** - ```bash - python -m sglang.launch_server --model $MODEL_PATH --tp 8 --ep 1 --port $PORT --quantization fp8 --kv-cache-dtype fp8_e4m3 - ``` - - **Valid format (arguments on separate lines):** - ```bash - python -m sglang.launch_server \ - --model $MODEL_PATH \ - --tp 8 \ - --ep 1 \ - --port $PORT \ - --quantization fp8 \ - --kv-cache-dtype fp8_e4m3 - ``` - - The same applies to `vllm serve` commands: - ```bash - vllm serve $MODEL_PATH \ - --tensor-parallel-size 8 \ - --port $PORT \ - --quantization fp8 - ``` - - If a benchmark script has server launch commands on a single line: - - This is a 🟡 **WARNING** issue - - Comment: "Server launch commands (`sglang.launch_server` or `vllm serve`) should have each argument on a separate line for better readability and easier code review. Please reformat using line continuations (`\`)." - - ### 2. MTP (Multi-Token Prediction) Benchmark Requirements: - When reviewing MTP benchmark scripts (files containing `mtp` in the name or using EAGLE speculative decoding): - - **Required: `--use-chat-template` flag** - MTP benchmarks MUST include the `--use-chat-template` flag in the benchmark client configuration. - - **Why it matters:** - - Chat templates ensure proper tokenization for speculative decoding - - Without this flag, MTP performance may be incorrect or suboptimal - - Ensures consistent behavior across different model configurations - - **Validation:** - Check that any benchmark script with MTP/EAGLE speculative decoding includes `--use-chat-template`: - ```bash - benchmark_client ... --use-chat-template - ``` - - If an MTP benchmark script is missing `--use-chat-template`: - - This is a 🔴 **BLOCKING** issue - - Comment: "MTP benchmark scripts MUST include `--use-chat-template` flag in the benchmark client configuration. This ensures proper tokenization for speculative decoding. Please add `--use-chat-template` to the benchmark command." - - ## Model Prefix Validation: - When reviewing changes to `configs/*-master.yaml` files, verify that ALL config keys use valid model prefixes. - - **Valid model prefixes:** - - `dsr1` - DeepSeek R1 models - - `gptoss` - GPT-OSS models - - **Invalid model prefixes (will break frontend):** - - `dsr1-fp8` - INVALID: precision should NOT be part of the model prefix - - `dsr1-fp4` - INVALID: precision should NOT be part of the model prefix - - Any other prefix not in the valid list above - - **Config key format:** - Config keys follow the pattern: `{model-prefix}-{precision}-{hardware}-{framework}` - Example valid keys: - - `dsr1-fp8-gb200-vllm` (model-prefix=dsr1, precision=fp8) - - `gptoss-fp4-h200-sglang` (model-prefix=gptoss, precision=fp4) - - **Why this matters:** - The frontend expects specific model prefixes (`dsr1` or `gptoss`) to display benchmark results correctly. Using invalid prefixes like `dsr1-fp8` will cause the frontend to fail to display results, breaking the user experience. - - **Validation:** - When reviewing config additions or changes, check that the first segment of config keys (before the first `-`) is either `dsr1` or `gptoss`. - - If a config key uses an invalid model prefix: - - This is a 🔴 **BLOCKING** issue - - Comment: "Invalid model prefix detected. The frontend only supports `dsr1` and `gptoss` as model prefixes. Using other prefixes like `dsr1-fp8` will break the frontend and prevent benchmark results from being displayed. Please use the format `{valid-prefix}-{precision}-{hardware}-{framework}` where valid-prefix is either `dsr1` or `gptoss`." diff --git a/.github/workflows/claude.yml b/.github/workflows/claude.yml index ed84f899b9..9b7414ab55 100644 --- a/.github/workflows/claude.yml +++ b/.github/workflows/claude.yml @@ -1,6 +1,9 @@ name: Claude Code -on: +# Distinct authorized comment requests must complete independently. +on: # zizmor: ignore[concurrency-limits] + pull_request: + types: [ready_for_review] issue_comment: types: [created] issues: @@ -8,23 +11,29 @@ on: pull_request_review_comment: types: [created] +permissions: + contents: read + jobs: claude: + name: claude if: | ((github.event_name == 'issue_comment' || github.event_name == 'pull_request_review_comment') && (contains(github.event.comment.body, '@claude') || contains(github.event.comment.body, '@Klaud-Cold'))) || (github.event_name == 'issues' && (contains(github.event.issue.body, '@claude') || contains(github.event.issue.title, '@claude') || contains(github.event.issue.body, '@Klaud-Cold') || contains(github.event.issue.title, '@Klaud-Cold'))) runs-on: ubuntu-latest permissions: - contents: write - pull-requests: write - issues: write - actions: read + contents: write # Let the authorized agent commit and push changes. + pull-requests: write # Publish PR feedback. + issues: write # Update comments, labels, and reactions. + actions: read # Read workflow runs and artifacts. steps: - name: Checkout repository uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: fetch-depth: 0 - token: ${{ secrets.CLAUDE_PAT }} + # Repository agent credential; the pinned action authorizes the triggering actor. + token: ${{ secrets.CLAUDE_PAT }} # zizmor: ignore[secrets-outside-env] + persist-credentials: false - name: Set up uv uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1 @@ -33,14 +42,18 @@ jobs: id: claude uses: anthropics/claude-code-action@0d0e0876d3eaa933f45dc692f7a4312c83caf36f # v1.0.218 env: - GH_TOKEN: ${{ secrets.CLAUDE_PAT }} - GITHUB_TOKEN: ${{ secrets.CLAUDE_PAT }} + # Repository agent credential; the pinned action authorizes the triggering actor. + GH_TOKEN: ${{ secrets.CLAUDE_PAT }} # zizmor: ignore[secrets-outside-env] + # Repository agent credential; the pinned action authorizes the triggering actor. + GITHUB_TOKEN: ${{ secrets.CLAUDE_PAT }} # zizmor: ignore[secrets-outside-env] INFERENCEMAX_ROOT: ${{ github.workspace }} BASH_DEFAULT_TIMEOUT_MS: "1800000" BASH_MAX_TIMEOUT_MS: "3600000" with: - anthropic_api_key: ${{ secrets.ANTHROPIC_API_KEY }} - github_token: ${{ secrets.CLAUDE_PAT }} + # Repository agent credential; the pinned action authorizes the triggering actor. + anthropic_api_key: ${{ secrets.ANTHROPIC_API_KEY }} # zizmor: ignore[secrets-outside-env] + # Repository agent credential; the pinned action authorizes the triggering actor. + github_token: ${{ secrets.CLAUDE_PAT }} # zizmor: ignore[secrets-outside-env] trigger_phrase: "${{ contains(github.event.comment.body || github.event.issue.body || github.event.issue.title || '', '@Klaud-Cold') && '@Klaud-Cold' || '@claude' }}" track_progress: true allowed_bots: '*' @@ -271,3 +284,328 @@ jobs: # Then use $EP in the vllm serve command ``` This ensures the script respects the `ep` setting in the master config YAML's search-space. + + review: + name: review + runs-on: ubuntu-latest + concurrency: + group: pr-review-${{ github.event.pull_request.number || github.event.issue.number }} + cancel-in-progress: false + # Only run if: + # 1. It's a PR event from someone with write access, OR + # 2. It's a comment containing @pr-claude from someone with write access + if: | + (github.event_name == 'pull_request' && + contains(fromJson('["OWNER", "MEMBER", "COLLABORATOR"]'), github.event.pull_request.author_association)) || + ((github.event_name == 'issue_comment' || github.event_name == 'pull_request_review_comment') && + contains(github.event.comment.body, '@pr-claude') && + contains(fromJson('["OWNER", "MEMBER", "COLLABORATOR"]'), github.event.comment.author_association)) + permissions: + contents: read + pull-requests: write # Publish PR feedback. + actions: read # Read workflow runs and artifacts. + id-token: write # Authenticate the Claude action with GitHub OIDC. + steps: + - name: Checkout repository + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + fetch-depth: 0 + persist-credentials: false + + - name: Set up uv + uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1 + + - name: PR Review with Claude + uses: anthropics/claude-code-action@0d0e0876d3eaa933f45dc692f7a4312c83caf36f # v1.0.218 + env: + INFERENCEMAX_ROOT: ${{ github.workspace }} + with: + # Repository review credential; the job gates eligible review requests. + anthropic_api_key: ${{ secrets.ANTHROPIC_API_KEY }} # zizmor: ignore[secrets-outside-env] + trigger_phrase: "@pr-claude" + track_progress: true + allowed_bots: '' + settings: | + {"fastMode": true} + + claude_args: | + --model 'claude-fable-5-1' + --mcp-config .github/mcp-ci.json + --allowedTools "mcp__github_inline_comment__create_inline_comment,mcp__inferencemax-repos__*,Bash(gh pr comment:*),Bash(gh pr diff:*),Bash(gh pr view:*)" + prompt: | + REPO: ${{ github.repository }} + PR NUMBER: ${{ github.event.pull_request.number || github.event.issue.number }} + + You are reviewing code for InferenceMAX. Your job is to provide HIGH-SIGNAL feedback only. + + ## Commands: + - `@pr-claude review` - Full review of the PR + - `@pr-claude re-review` - Re-review only NEW changes since your last review (check your previous comments first) + - `@pr-claude review ` - Review only a specific file + - `@pr-claude` followed by a question - Answer the question about this PR + + ## If this is a re-review: + 1. First, check existing review comments on this PR using `gh pr view` + 2. Focus ONLY on new commits or changes not previously reviewed + 3. Do NOT repeat previous feedback - reference it if still applicable + 4. If previous issues were fixed, acknowledge briefly in the summary + + ## ONLY comment when you find: + 1. **Bugs**: Code that is broken, will crash, or produces incorrect results + 2. **Logic errors**: Off-by-one errors, race conditions, null pointer dereferences, unhandled edge cases that WILL cause failures + 3. **Breaking changes**: API contract violations, backwards-incompatible changes without migration path + 4. **Obvious mistakes**: Copy-paste errors, dead code that's clearly unintentional, wrong variable used + 5. **Resource leaks**: Unclosed connections, missing cleanup, memory leaks + + ## DO NOT comment on: + - Style preferences or formatting (we have linters for that) + - "Consider doing X" suggestions unless the current code is actually broken + - Minor naming nitpicks + - Adding more comments or documentation + - Theoretical performance improvements without evidence of actual impact + - "Best practices" that don't apply to this specific context + - Praise or positive feedback (save it for the summary) + - Issues you already commented on in a previous review + + ## Comment format: + For each issue, use inline comments with this format: + **[SEVERITY]**: Brief description of the actual problem + **Why it matters**: What will break or go wrong + **Fix**: Concrete suggestion (not vague advice) + **Fix** When possible, the fix should use the GitHub Multi line Code Suggestion. Here is an example on how to use it + + start with 3 backtick symbols followed by the keyword suggestion and then for lines to suggest to delete use the minus symbol and for lines to suggest add, use the plus symbol + + ```suggestion + - line to delete + + line to add + ``` + + Severity levels: + - 🔴 **BLOCKING**: Must fix before merge - will cause bugs/crashes/security issues + - 🟡 **WARNING**: Should fix - likely to cause problems in edge cases + + ## Output: + - Use `mcp__github_inline_comment__create_inline_comment` for specific code issues + - Use `gh pr comment` ONCE at the end for a brief summary (max 3-4 sentences) + - If the PR looks good with no issues, just say "LGTM - no blocking issues found" and nothing else + - For re-reviews, prefix summary with "Re-review:" and note what changed + + ## Master Config and Perf Changelog Validation: + When reviewing a PR, check if any of the following master config files were modified: + - `configs/amd-master.yaml` + - `configs/nvidia-master.yaml` + + If either master config file was edited AND `perf-changelog.yaml` was NOT edited in the same PR: + - This is a 🔴 **BLOCKING** issue + - Comment that `perf-changelog.yaml` must also be updated when master config files are changed + - The perf-changelog entry should document what changed in the config and include the PR link + - Format: "Master config files were modified but `perf-changelog.yaml` was not updated. When changing `configs/amd-master.yaml` or `configs/nvidia-master.yaml`, you must add a corresponding entry to `perf-changelog.yaml` documenting the changes." + + ### Perf Changelog Entry Position: + `perf-changelog.yaml` is read in chronological order — oldest entries at the top, newest at the bottom. New entries MUST be appended to the END of the file. Never insert in the middle or prepend. + + When reviewing a PR diff for `perf-changelog.yaml`: + - Check that every newly added entry lands at the very end of the file (i.e., the diff hunk adds lines after the previous last entry, not before existing entries). + - If any new entry is inserted above/between existing entries rather than appended to the end: + - This is a 🔴 **BLOCKING** issue + - Comment: "New `perf-changelog.yaml` entries must be appended to the END of the file. The file is read chronologically (oldest at top, newest at bottom), so inserting in the middle or prepending breaks the ordering. Please move the new entry(ies) to the bottom of the file." + + ### Append-only Perf Changelog Safety: + When a new `perf-changelog.yaml` entry contains `append-only: true`, verify the complete PR diff before approving it: + - Do not use a file allowlist. Supporting code, benchmark scripts, launchers, helpers, and other files may change. Inspect the complete diff and judge whether each benchmark-affecting change is behaviorally isolated to the appended points. + - Every newly added changelog entry must contain `append-only: true`; append-only and regular entries may not be mixed. + - Treat the generated base matrix as an immutable subset of the generated head matrix. Every existing point must remain present with the same image and complete recipe. Additions may include new concurrency values or entirely new recipe variants (for example, a new tensor-parallelism value) inside the selected existing config/scenario, but they must retain the existing visual curve's single non-null image. + - Benchmark or launch logic may change only when every changed behavior is on a control-flow path uniquely gated to the corresponding newly appended points. Trace the selected config and generated runtime values through every affected file into the condition. Confirm the path cannot be reached by any existing point. Unguarded/shared setup changes, or a branch also used by an existing concurrency/config/scenario, are blocking. + - No existing point, recipe variant, config, or scenario may be removed or replaced. New configs and scenarios are out of scope for append-only mode; new generated variants inside the selected existing config/scenario are allowed. + - Eval modifiers (`evals-only`, `all-evals`, `eval-min-prefill-ep`) are not allowed. + If any condition fails, report a 🔴 **BLOCKING** issue. Never reject a change merely because of its file path; reject it when its benchmark effect is not exclusive to the appended points or the exclusivity cannot be proven from the diff. + + ## Terminology: + - **STP (Single Token Prediction)**: Standard autoregressive decoding — one token per forward pass. No speculative decoding or MTP. Benchmarks labeled "STP only" use vanilla decoding. + - **MTP (Multi-Token Prediction)**: Predicts multiple tokens per forward pass using speculative decoding (e.g., EAGLE, NEXTN). + + ## Expert Parallelism Validation: + When reviewing benchmark scripts for MoE models, verify that expert parallelism flags are used correctly: + + - **vLLM** (`vllm serve`): Uses `--enable-expert-parallel` (boolean flag). Does NOT accept `--expert-parallel-size`. EP size is automatically determined by vLLM. + - **SGLang** (`sglang.launch_server`): Uses `--expert-parallel-size N` (explicit integer). + - **ATOM** (AMD vLLM fork): Uses `--enable-expert-parallel` (same as vLLM). + + **Required pattern for vLLM/ATOM scripts:** + Scripts must NOT hardcode `--enable-expert-parallel`. Instead, they should conditionally enable it based on the `EP_SIZE` env var: + ```bash + if [ "$EP_SIZE" -gt 1 ]; then + EP=" --enable-expert-parallel" + else + EP=" " + fi + ``` + If a script hardcodes `--enable-expert-parallel` without checking `EP_SIZE`: + - This is a 🟡 **WARNING** issue + - Comment: "Expert parallelism should be conditional on the `EP_SIZE` env var from the config YAML, not hardcoded. Use the `if [ \"$EP_SIZE\" -gt 1 ]` pattern to conditionally enable `--enable-expert-parallel`." + + Remember: Silence is golden. No comment is better than a low-value comment. + + ## Container Image Accessibility Validation: + When reviewing changes to `configs/*-master.yaml` files, verify that ALL `image:` values are publicly accessible: + + **Valid image formats (publicly accessible):** + - Docker Hub: `organization/image:tag` (e.g., `lmsysorg/sglang:v0.5.7-rocm700-mi35x`) + - NGC: `nvcr.io/nvidia/...` or `nvcr.io#nvidia/...` (e.g., `nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post1` or `nvcr.io#nvidia/tensorrt-llm/release:1.1.0rc2.post2`) + - Other public registries: `ghcr.io/...`, `quay.io/...`, `rocm/...` + + **Invalid image formats (NOT publicly accessible):** + - generally these images are not best practices to have in: + - Local file paths: `/scratch/...`, `/home/...`, `/data/...`, or any path starting with `/` + - `.sqsh` files (squashfs containers stored locally) + - Internal/private registry paths that are not publicly resolvable + + If any `image:` field contains a local path or non-public image: + - This is a 🔴 **BLOCKING** issue + - Comment: "Image must be publicly accessible on NGC, Docker Hub, or another public registry. Local paths like `/scratch/...` or `.sqsh` files are generally not accepted. Please push the container to a public registry (e.g., `nvcr.io/nvidia/...` for NGC) and update the config with the public image reference." + - Link to the specific line with the invalid image path + + ## Enroot Import Validation for Launch Scripts: + When reviewing changes to `runners/launch_*.sh` files, verify that the script properly transforms public Docker images to enroot local images for reproducibility. + + **Expected pattern:** + The script should include an enroot import command like: + ```bash + srun --jobid=$JOB_ID bash -c "enroot import -o $SQUASH_FILE docker://$IMAGE" + ``` + or similar variations such as: + ```bash + srun -N 1 -A $SLURM_ACCOUNT -p $SLURM_PARTITION bash -c "enroot import -o $SQUASH_FILE docker://$IMAGE" + ``` + + **Why this matters:** + - Ensures the exact same public NGC/Docker Hub image is used + - Makes benchmarks reproducible by anyone with access to the public image + - Prevents reliance on pre-existing local container images that others cannot access + + **Validation Steps:** + 1. Look for `enroot import` commands that convert `docker://` images to local `.sqsh` files + 2. The image source should be a public registry (NGC, Docker Hub, etc.), not a local path + 3. If the script uses containers but does NOT have an `enroot import docker://` pattern: + - This is a 🟡 **WARNING** issue + - Comment: "This launch script uses container images but does not appear to transform a public Docker image to an enroot local image using `enroot import -o $SQUASH_FILE docker://$IMAGE`. For reproducibility, please either: + 1. Add the enroot import pattern to pull from a public registry, OR + 2. Explain why this script has a different workflow (e.g., uses a different container runtime, pre-built images are acceptable for this use case, etc.)" + - Ask the developer to provide a reasonable explanation if the pattern is intentionally omitted + + ## vLLM and SGLang Source Code Access: + You have access to vLLM and SGLang source code via the inferencemax-repos MCP server: + - Use `mcp__inferencemax-repos__*` tools to access repository source code + - Resources are available via URIs: `vllm:///path/to/file.py` and `sglang:///path/to/file.py` + - The server automatically detects and checks out the version matching InferenceMAX configs + - Use the `list_versions` tool to see detected versions + - Use the `switch_version` tool to switch to a different version if needed + + **When to use this:** + - When reviewing changes that interact with vLLM/SGLang APIs or internals + - To verify that code correctly uses vLLM/SGLang features + - To check if assumptions about vLLM/SGLang behavior are accurate + - To understand how vLLM/SGLang implements features being used/modified + - To cross-reference documentation claims with actual implementation + + **Example use cases:** + - PR modifies vLLM scheduler usage → Check actual vLLM scheduler implementation + - PR adds SGLang integration → Verify SGLang API matches the PR's usage + - PR claims feature X works a certain way → Read the source to confirm + - PR has a bug related to KV cache → Look at vLLM's KV cache implementation + + ## Line Count Report for the matrix generator: + If `infx/matrix/generate.py` was modified in this PR: + 1. Count the total lines in the current (PR) version of the file + 2. Count the lines in the base branch version of `infx/matrix/generate.py`. If that path does not exist in the base revision, use `utils/matrix_logic/generate_sweep_configs.py` from that revision instead. This fallback compares the implementation across the package migration rather than counting the new path as entirely new code. + 3. Calculate the difference (current - base) + 4. Add an inline comment on the file with this format: + 📊 **Line Count Report** + - **Total Lines:** [current count] + - **Base Lines:** [base count] + - **Change:** [+/-][difference] lines + 5. Use 📈 for additions, 📉 for removals, ➡️ for no change + + ## Benchmark Script Code Style Guidelines: + When reviewing changes to `benchmarks/*.sh` files, verify the following code style requirements: + + ### 1. Server Launch Command Formatting: + All `sglang.launch_server` and `vllm serve` commands MUST have their arguments formatted on separate lines for readability. + + **Invalid format (all arguments on one line):** + ```bash + python -m sglang.launch_server --model $MODEL_PATH --tp 8 --ep 1 --port $PORT --quantization fp8 --kv-cache-dtype fp8_e4m3 + ``` + + **Valid format (arguments on separate lines):** + ```bash + python -m sglang.launch_server \ + --model $MODEL_PATH \ + --tp 8 \ + --ep 1 \ + --port $PORT \ + --quantization fp8 \ + --kv-cache-dtype fp8_e4m3 + ``` + + The same applies to `vllm serve` commands: + ```bash + vllm serve $MODEL_PATH \ + --tensor-parallel-size 8 \ + --port $PORT \ + --quantization fp8 + ``` + + If a benchmark script has server launch commands on a single line: + - This is a 🟡 **WARNING** issue + - Comment: "Server launch commands (`sglang.launch_server` or `vllm serve`) should have each argument on a separate line for better readability and easier code review. Please reformat using line continuations (`\`)." + + ### 2. MTP (Multi-Token Prediction) Benchmark Requirements: + When reviewing MTP benchmark scripts (files containing `mtp` in the name or using EAGLE speculative decoding): + + **Required: `--use-chat-template` flag** + MTP benchmarks MUST include the `--use-chat-template` flag in the benchmark client configuration. + + **Why it matters:** + - Chat templates ensure proper tokenization for speculative decoding + - Without this flag, MTP performance may be incorrect or suboptimal + - Ensures consistent behavior across different model configurations + + **Validation:** + Check that any benchmark script with MTP/EAGLE speculative decoding includes `--use-chat-template`: + ```bash + benchmark_client ... --use-chat-template + ``` + + If an MTP benchmark script is missing `--use-chat-template`: + - This is a 🔴 **BLOCKING** issue + - Comment: "MTP benchmark scripts MUST include `--use-chat-template` flag in the benchmark client configuration. This ensures proper tokenization for speculative decoding. Please add `--use-chat-template` to the benchmark command." + + ## Model Prefix Validation: + When reviewing changes to `configs/*-master.yaml` files, verify that ALL config keys use valid model prefixes. + + **Valid model prefixes:** + - `dsr1` - DeepSeek R1 models + - `gptoss` - GPT-OSS models + + **Invalid model prefixes (will break frontend):** + - `dsr1-fp8` - INVALID: precision should NOT be part of the model prefix + - `dsr1-fp4` - INVALID: precision should NOT be part of the model prefix + - Any other prefix not in the valid list above + + **Config key format:** + Config keys follow the pattern: `{model-prefix}-{precision}-{hardware}-{framework}` + Example valid keys: + - `dsr1-fp8-gb200-vllm` (model-prefix=dsr1, precision=fp8) + - `gptoss-fp4-h200-sglang` (model-prefix=gptoss, precision=fp4) + + **Why this matters:** + The frontend expects specific model prefixes (`dsr1` or `gptoss`) to display benchmark results correctly. Using invalid prefixes like `dsr1-fp8` will cause the frontend to fail to display results, breaking the user experience. + + **Validation:** + When reviewing config additions or changes, check that the first segment of config keys (before the first `-`) is either `dsr1` or `gptoss`. + + If a config key uses an invalid model prefix: + - This is a 🔴 **BLOCKING** issue + - Comment: "Invalid model prefix detected. The frontend only supports `dsr1` and `gptoss` as model prefixes. Using other prefixes like `dsr1-fp8` will break the frontend and prevent benchmark results from being displayed. Please use the format `{valid-prefix}-{precision}-{hardware}-{framework}` where valid-prefix is either `dsr1` or `gptoss`." diff --git a/.github/workflows/codeowner-signoff-verify.yml b/.github/workflows/codeowner-signoff-verify.yml index a1316e67bb..e8380a1378 100644 --- a/.github/workflows/codeowner-signoff-verify.yml +++ b/.github/workflows/codeowner-signoff-verify.yml @@ -1,7 +1,8 @@ name: CODEOWNER Sign-off Verify on: - pull_request_target: + # Fork sign-off needs a trusted status writer; only default-branch code executes and verifier authorization is checked. + pull_request_target: # zizmor: ignore[dangerous-triggers] types: [opened, synchronize, reopened, ready_for_review] issue_comment: types: [created, edited, deleted] @@ -45,10 +46,10 @@ jobs: cancel-in-progress: false permissions: contents: read - pull-requests: write - issues: write - actions: read - statuses: write + pull-requests: write # Publish PR feedback. + issues: write # Update comments, labels, and reactions. + actions: read # Read workflow runs and artifacts. + statuses: write # Publish the CODEOWNER status. steps: - name: Checkout trusted workflow code uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 @@ -281,14 +282,18 @@ jobs: if: steps.prepare.outputs.verify == 'true' uses: anthropics/claude-code-action@0d0e0876d3eaa933f45dc692f7a4312c83caf36f # v1.0.218 env: - GH_TOKEN: ${{ secrets.CLAUDE_PAT }} - GITHUB_TOKEN: ${{ secrets.CLAUDE_PAT }} + # Repository verifier credential; trusted code checks writer authorization before this step. + GH_TOKEN: ${{ secrets.CLAUDE_PAT }} # zizmor: ignore[secrets-outside-env] + # Repository verifier credential; trusted code checks writer authorization before this step. + GITHUB_TOKEN: ${{ secrets.CLAUDE_PAT }} # zizmor: ignore[secrets-outside-env] INFERENCEMAX_ROOT: ${{ github.workspace }} BASH_DEFAULT_TIMEOUT_MS: "1800000" BASH_MAX_TIMEOUT_MS: "3600000" with: - anthropic_api_key: ${{ secrets.ANTHROPIC_API_KEY }} - github_token: ${{ secrets.CLAUDE_PAT }} + # Repository verifier credential; trusted code checks writer authorization before this step. + anthropic_api_key: ${{ secrets.ANTHROPIC_API_KEY }} # zizmor: ignore[secrets-outside-env] + # Repository verifier credential; trusted code checks writer authorization before this step. + github_token: ${{ secrets.CLAUDE_PAT }} # zizmor: ignore[secrets-outside-env] track_progress: false allowed_bots: '' additional_permissions: | diff --git a/.github/workflows/collect-evals.yml b/.github/workflows/collect-evals.yml index a320717bec..f0c78b7425 100644 --- a/.github/workflows/collect-evals.yml +++ b/.github/workflows/collect-evals.yml @@ -13,13 +13,14 @@ permissions: jobs: collect-evals: + name: collect-evals runs-on: ubuntu-latest steps: - name: Checkout code uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: - token: ${{ secrets.REPO_PAT }} fetch-depth: 0 + persist-credentials: false - name: Download eval artifacts uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 @@ -31,11 +32,13 @@ jobs: uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1 - name: Summarize evals + env: + RESULT_PREFIX: ${{ inputs.result-prefix || 'all' }} run: | echo "## Eval Summary" >> $GITHUB_STEP_SUMMARY echo "" >> $GITHUB_STEP_SUMMARY uv run --no-project --exclude-newer PT12H --python 3.12 --with tabulate \ - python -m infx.results.collect_eval_results eval_results/ ${{ inputs.result-prefix || 'all' }} >> $GITHUB_STEP_SUMMARY + python -m infx.results.collect_eval_results eval_results/ "${RESULT_PREFIX}" >> $GITHUB_STEP_SUMMARY - name: Upload aggregated evals uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 diff --git a/.github/workflows/collect-results.yml b/.github/workflows/collect-results.yml index 5da93d221e..5b06d4eeb6 100644 --- a/.github/workflows/collect-results.yml +++ b/.github/workflows/collect-results.yml @@ -13,14 +13,15 @@ permissions: jobs: collect-results: + name: collect-results runs-on: ubuntu-latest steps: - name: Checkout code uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: - token: ${{ secrets.REPO_PAT }} fetch-depth: 0 + persist-credentials: false - name: Download JSON artifacts uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 @@ -29,7 +30,10 @@ jobs: pattern: ${{ inputs.result-prefix && format('{0}_*', inputs.result-prefix) || '*' }} - name: Aggregate results - run: python3 -m infx.results.collect_results results/ ${{ inputs.result-prefix || 'all' }} + env: + RESULT_PREFIX: ${{ inputs.result-prefix || 'all' }} + run: | + python3 -m infx.results.collect_results results/ "${RESULT_PREFIX}" - name: Upload aggregated results uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 diff --git a/.github/workflows/collectivex-sweep.yml b/.github/workflows/collectivex-sweep.yml index 379975a497..06bd475284 100644 --- a/.github/workflows/collectivex-sweep.yml +++ b/.github/workflows/collectivex-sweep.yml @@ -1,7 +1,7 @@ # Generate a shard matrix, run each shard on its GPU pool, and upload raw results. name: CollectiveX Sweep permissions: - actions: read + actions: read # Read workflow runs and artifacts. contents: read on: workflow_dispatch: @@ -32,10 +32,10 @@ on: type: boolean default: false fp8_consume: - description: "DIAGNOSTIC ONLY. 'dequant' puts the FP8 dequant back inside the chain, so staging is no longer hoisted; changes what is measured. Blank = native (the production model)." + description: "DIAGNOSTIC ONLY. 'dequant' puts the FP8 dequant back inside the chain, so staging is no longer hoisted; changes what is measured. Use native for the production model." type: choice - default: '' - options: ['', dequant] + default: native + options: [native, dequant] concurrency: group: cx-${{ github.ref }}-${{ inputs.backend }}-${{ inputs.only_sku }} cancel-in-progress: false @@ -43,6 +43,7 @@ concurrency: jobs: # ---- setup: generate the shard matrix and upload its requested coverage ---- setup: + name: setup runs-on: ubuntu-latest outputs: matrix: ${{ steps.gen.outputs.matrix }} @@ -223,17 +224,22 @@ jobs: path: ${{ env.COLLX_JOB_ROOT }}/control # Materialize this shard's .shards/.json case control from the run-scoped matrix artifact downloaded above. - name: Extract execution control + env: + MATRIX_ID: ${{ matrix.id }} run: | set -eo pipefail cd "$COLLX_SOURCE_ROOT/experimental/CollectiveX" 2>/dev/null \ || { echo "CollectiveX source is unavailable" >&2; exit 1; } python3 sweep_matrix.py \ --extract-from "$COLLX_JOB_ROOT/control/matrix_full.json" \ - --shard-id '${{ matrix.id }}' \ - --out '${{ env.COLLX_SHARD_FILE }}' >/dev/null + --shard-id "${MATRIX_ID}" \ + --out "${COLLX_SHARD_FILE}" >/dev/null + # Merge the base + per-SKU operator-config secrets into one validated mode-0600 file, then run the SKU launcher (allocate -> stage -> build -> run cases). - name: Execute sweep cell ${{ matrix.id }} id: sweep_shard + env: + MATRIX_LAUNCHER: ${{ matrix.launcher }} run: | set -eo pipefail umask 077 @@ -242,7 +248,8 @@ jobs: # launcher's collx_load_operator_config emits it per SKU. cd "$COLLX_SOURCE_ROOT" 2>/dev/null \ || { echo "CollectiveX source is unavailable" >&2; exit 1; } - bash "experimental/CollectiveX/launchers/launch_${{ matrix.launcher }}.sh" + bash "experimental/CollectiveX/launchers/launch_${MATRIX_LAUNCHER}.sh" + # always(): cancel any Slurm allocation a killed launcher left recorded, append the summary table, and stage result JSONs so a red or partial leg still uploads. - name: Summarize and stage shard results id: stage_artifact diff --git a/.github/workflows/e2e-tests.yml b/.github/workflows/e2e-tests.yml index 8b824765c4..5f11e84b39 100644 --- a/.github/workflows/e2e-tests.yml +++ b/.github/workflows/e2e-tests.yml @@ -4,7 +4,8 @@ run-name: e2e Test - ${{ inputs.test-name || inputs.generate-cli-command || gith permissions: contents: read -on: +# Independent GPU dispatches queue through the priority scheduler; newer runs must not cancel them. +on: # zizmor: ignore[concurrency-limits] workflow_dispatch: inputs: generate-cli-command: @@ -101,6 +102,15 @@ on: type: string default: "[]" workflow_call: + secrets: + REPO_PAT: + required: true + INFERENCEX_OFFICIAL_RO_HF_TOKEN: + required: true + MODAL_TOKEN_ID: + required: false + MODAL_TOKEN_SECRET: + required: false inputs: generate-cli-command: description: "Command passed to generate matrix script" @@ -198,6 +208,7 @@ on: jobs: get-jobs: + name: get-jobs runs-on: ubuntu-latest outputs: single-node-config: ${{ steps.get-jobs.outputs.single-node-config }} @@ -222,6 +233,7 @@ jobs: uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: ref: ${{ github.sha }} + persist-credentials: false - name: Checkout workflow tooling if: ${{ inputs.ref && inputs.ref != '' }} uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 @@ -233,6 +245,7 @@ jobs: - uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1 - id: get-jobs env: + GENERATE_COMMAND: ${{ inputs.generate-cli-command || github.event.inputs.generate-cli-command }} PR_LABELS: ${{ inputs.pr-labels-json || toJson(github.event.pull_request.labels.*.name) }} CHANGELOG_BASE_REF: ${{ inputs.changelog-base-ref }} CHANGELOG_HEAD_REF: ${{ inputs.changelog-head-ref }} @@ -269,7 +282,6 @@ jobs: RAW_CONFIG_JSON=$("${CMD[@]}") CONFIG_JSON=$(python3 -c 'import json,sys; data=json.load(sys.stdin); rows=[row for family in ("single_node","multi_node") for group in data.get(family,{}).values() for row in group]; rows.extend(row for family in ("evals","agentic_evals","multinode_evals","multinode_agentic_evals") for row in data.get(family,[])); print(json.dumps(rows))' <<<"$RAW_CONFIG_JSON") else - GENERATE_COMMAND="${{ inputs.generate-cli-command || github.event.inputs.generate-cli-command }}" if [ -z "$GENERATE_COMMAND" ]; then echo "generate-cli-command is required outside trusted changelog dispatch mode" >&2 exit 1 @@ -296,8 +308,8 @@ jobs: env PYTHONPATH="$PRIORITY_ROOT" uv run --no-project --exclude-newer PT12H --python 3.12 --with pyyaml \ python -P "${PRIORITY_ROOT}/utils/ci_priority.py" \ --policy "${PRIORITY_ROOT}/configs/ci-priority.yaml" \ - --event-name "${{ github.event_name }}" \ - --queue-namespace "${{ github.run_id }}:${{ github.run_attempt }}:${family}" \ + --event-name "${GITHUB_EVENT_NAME}" \ + --queue-namespace "${GITHUB_RUN_ID}:${GITHUB_RUN_ATTEMPT}:${family}" \ --labels-json "$PR_LABELS" } AGENTIC=$(echo "$CONFIG_JSON" | python3 -c "import sys,json; d=json.load(sys.stdin); print(json.dumps([x for x in d if x.get('scenario-type') == 'agentic-coding' and 'prefill' not in x and not x.get('eval-only', False)]))" | score_matrix agentic) @@ -322,13 +334,17 @@ jobs: test-sweep-multi-node: needs: get-jobs if: ${{ needs.get-jobs.outputs.multi-node-config != '' && needs.get-jobs.outputs.multi-node-config != '[]' }} - uses: ./.github/workflows/benchmark-multinode-tmpl.yml + uses: $/.github/workflows/benchmark-multinode-tmpl.yml name: multi-node / strategy: fail-fast: ${{ inputs.fail-fast }} matrix: config: ${{ fromJson(needs.get-jobs.outputs.multi-node-config) }} - secrets: inherit + secrets: + REPO_PAT: ${{ secrets.REPO_PAT }} + INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} + MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} + MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} klaud-run: ${{ inputs.klaud-run }} @@ -345,13 +361,17 @@ jobs: test-sweep-multi-node-evals: needs: get-jobs if: ${{ needs.get-jobs.outputs.multi-node-eval-config != '' && needs.get-jobs.outputs.multi-node-eval-config != '[]' }} - uses: ./.github/workflows/benchmark-multinode-tmpl.yml + uses: $/.github/workflows/benchmark-multinode-tmpl.yml name: multi-node eval / strategy: fail-fast: ${{ inputs.fail-fast }} matrix: config: ${{ fromJson(needs.get-jobs.outputs.multi-node-eval-config) }} - secrets: inherit + secrets: + REPO_PAT: ${{ secrets.REPO_PAT }} + INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} + MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} + MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} klaud-run: ${{ inputs.klaud-run }} @@ -372,13 +392,17 @@ jobs: test-sweep-agentic: needs: get-jobs if: ${{ needs.get-jobs.outputs.agentic-config != '' && needs.get-jobs.outputs.agentic-config != '[]' }} - uses: ./.github/workflows/benchmark-tmpl.yml + uses: $/.github/workflows/benchmark-tmpl.yml name: agentic / strategy: fail-fast: ${{ inputs.fail-fast }} matrix: config: ${{ fromJson(needs.get-jobs.outputs.agentic-config) }} - secrets: inherit + secrets: + REPO_PAT: ${{ secrets.REPO_PAT }} + INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} + MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} + MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} klaud-run: ${{ inputs.klaud-run }} @@ -396,13 +420,17 @@ jobs: test-sweep-agentic-evals: needs: get-jobs if: ${{ needs.get-jobs.outputs.agentic-eval-config != '' && needs.get-jobs.outputs.agentic-eval-config != '[]' }} - uses: ./.github/workflows/benchmark-tmpl.yml + uses: $/.github/workflows/benchmark-tmpl.yml name: agentic eval / strategy: fail-fast: ${{ inputs.fail-fast }} matrix: config: ${{ fromJson(needs.get-jobs.outputs.agentic-eval-config) }} - secrets: inherit + secrets: + REPO_PAT: ${{ secrets.REPO_PAT }} + INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} + MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} + MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} klaud-run: ${{ inputs.klaud-run }} @@ -424,13 +452,17 @@ jobs: test-sweep-multi-node-agentic: needs: get-jobs if: ${{ needs.get-jobs.outputs.multi-node-agentic-config != '' && needs.get-jobs.outputs.multi-node-agentic-config != '[]' }} - uses: ./.github/workflows/benchmark-multinode-tmpl.yml + uses: $/.github/workflows/benchmark-multinode-tmpl.yml name: multi-node agentic / strategy: fail-fast: ${{ inputs.fail-fast }} matrix: config: ${{ fromJson(needs.get-jobs.outputs.multi-node-agentic-config) }} - secrets: inherit + secrets: + REPO_PAT: ${{ secrets.REPO_PAT }} + INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} + MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} + MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} klaud-run: ${{ inputs.klaud-run }} @@ -451,13 +483,17 @@ jobs: test-sweep-multi-node-agentic-evals: needs: get-jobs if: ${{ needs.get-jobs.outputs.multi-node-agentic-eval-config != '' && needs.get-jobs.outputs.multi-node-agentic-eval-config != '[]' }} - uses: ./.github/workflows/benchmark-multinode-tmpl.yml + uses: $/.github/workflows/benchmark-multinode-tmpl.yml name: multi-node agentic eval / strategy: fail-fast: false matrix: config: ${{ fromJson(needs.get-jobs.outputs.multi-node-agentic-eval-config) }} - secrets: inherit + secrets: + REPO_PAT: ${{ secrets.REPO_PAT }} + INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} + MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} + MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} klaud-run: ${{ inputs.klaud-run }} @@ -481,13 +517,17 @@ jobs: test-sweep-single-node: needs: get-jobs if: ${{ needs.get-jobs.outputs.single-node-config != '' && needs.get-jobs.outputs.single-node-config != '[]' }} - uses: ./.github/workflows/benchmark-tmpl.yml + uses: $/.github/workflows/benchmark-tmpl.yml name: single-node / strategy: fail-fast: ${{ inputs.fail-fast }} matrix: config: ${{ fromJson(needs.get-jobs.outputs.single-node-config) }} - secrets: inherit + secrets: + REPO_PAT: ${{ secrets.REPO_PAT }} + INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} + MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} + MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} klaud-run: ${{ inputs.klaud-run }} @@ -502,13 +542,17 @@ jobs: test-sweep-evals: needs: get-jobs if: ${{ needs.get-jobs.outputs.eval-config != '' && needs.get-jobs.outputs.eval-config != '[]' }} - uses: ./.github/workflows/benchmark-tmpl.yml + uses: $/.github/workflows/benchmark-tmpl.yml name: eval / strategy: fail-fast: ${{ inputs.fail-fast }} matrix: config: ${{ fromJson(needs.get-jobs.outputs.eval-config) }} - secrets: inherit + secrets: + REPO_PAT: ${{ secrets.REPO_PAT }} + INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} + MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} + MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} klaud-run: ${{ inputs.klaud-run }} @@ -525,20 +569,21 @@ jobs: ref: ${{ inputs.ref }} collect-results: + name: collect-results needs: [test-sweep-multi-node, test-sweep-single-node, test-sweep-agentic, test-sweep-multi-node-agentic] if: ${{ always() && (needs.test-sweep-multi-node.result != 'skipped' || needs.test-sweep-single-node.result != 'skipped' || needs.test-sweep-agentic.result != 'skipped' || needs.test-sweep-multi-node-agentic.result != 'skipped') }} - uses: ./.github/workflows/collect-results.yml - secrets: inherit + uses: $/.github/workflows/collect-results.yml with: result-prefix: "bmk" collect-evals: + name: collect-evals needs: [test-sweep-evals, test-sweep-multi-node-evals, test-sweep-agentic-evals, test-sweep-multi-node-agentic-evals] if: ${{ always() && (needs.test-sweep-evals.result != 'skipped' || needs.test-sweep-multi-node-evals.result != 'skipped' || needs.test-sweep-agentic-evals.result != 'skipped' || needs.test-sweep-multi-node-agentic-evals.result != 'skipped') }} - uses: ./.github/workflows/collect-evals.yml - secrets: inherit + uses: $/.github/workflows/collect-evals.yml calc-success-rate: + name: calc-success-rate needs: [collect-results, collect-evals] if: ${{ always() }} runs-on: ubuntu-latest @@ -553,6 +598,7 @@ jobs: with: token: ${{ secrets.REPO_PAT }} fetch-depth: 0 + persist-credentials: false - name: Download results artifacts uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 diff --git a/.github/workflows/klaud-candidate.yml b/.github/workflows/klaud-candidate.yml index ac23f60530..2ce336e31c 100644 --- a/.github/workflows/klaud-candidate.yml +++ b/.github/workflows/klaud-candidate.yml @@ -16,8 +16,8 @@ on: permissions: contents: read - actions: read - pull-requests: read + actions: read # Read workflow runs and artifacts. + pull-requests: read # Read PR metadata. jobs: claude: @@ -30,14 +30,14 @@ jobs: KLAUD_BRANCH: klaud/auto-${{ inputs.candidate }} KLAUD_TEST_NAME: klaud-${{ github.run_id }}-${{ inputs.candidate }} steps: - - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: persist-credentials: false fetch-depth: 0 - - uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d + - uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1 with: python-version: '3.12' - - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c + - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 with: name: klaud-plan path: ${{ runner.temp }}/plan @@ -100,7 +100,7 @@ jobs: --execution-file "${KLAUD_EXECUTION:-$RUNNER_TEMP/claude-execution-output.json}" \ --structured-outcome-file "$RUNNER_TEMP/candidate-outcome.json" \ --outcome "${KLAUD_OUTCOME:-unknown}" --output "$RUNNER_TEMP/candidate-diagnostics.json" - - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a + - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 if: always() with: name: klaud-candidate-${{ inputs.candidate }} diff --git a/.github/workflows/klaud-plan.yml b/.github/workflows/klaud-plan.yml index ba8eec4273..063b9679cb 100644 --- a/.github/workflows/klaud-plan.yml +++ b/.github/workflows/klaud-plan.yml @@ -1,14 +1,15 @@ name: Klaud Cold auto-sweep -on: +# Overlapping waves are intentional; candidate ownership claims prevent duplicate work. +on: # zizmor: ignore[concurrency-limits] workflow_dispatch: schedule: - cron: '0 */6 * * *' permissions: contents: read - actions: read - pull-requests: read + actions: read # Read workflow runs and artifacts. + pull-requests: read # Read PR metadata. jobs: plan: @@ -17,8 +18,8 @@ jobs: runs-on: ubuntu-latest permissions: contents: read - actions: read - pull-requests: read + actions: read # Read workflow runs and artifacts. + pull-requests: read # Read PR metadata. env: # Total candidates per invocation; selected candidates run in parallel. MAX_CANDIDATES_PER_RUN: '5' @@ -26,23 +27,25 @@ jobs: selected: ${{ steps.select.outputs.selected }} candidates: ${{ steps.select.outputs.candidates }} steps: - - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: persist-credentials: false fetch-depth: 0 - - uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d + - uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1 with: python-version: '3.12' - name: Reconcile interrupted owned candidates env: # Recovery mutates only sessions recorded by this workflow after their parent finishes. - GH_TOKEN: ${{ secrets.CLAUDE_PAT }} + # Repository automation credential; the planner runs only on main. + GH_TOKEN: ${{ secrets.CLAUDE_PAT }} # zizmor: ignore[secrets-outside-env] run: uv run --no-project --exclude-newer PT12H --python 3.12 --with "pydantic>=2.10,<3" --with pyyaml python -m infx.klaud recover - name: Prepare eligible candidates and open PRs id: prepare env: GH_TOKEN: ${{ github.token }} - KLAUD_DASHBOARD_API_KEY: ${{ secrets.KLAUD_DASHBOARD_API_KEY }} + # Repository automation credential; the planner runs only on main. + KLAUD_DASHBOARD_API_KEY: ${{ secrets.KLAUD_DASHBOARD_API_KEY }} # zizmor: ignore[secrets-outside-env] run: uv run --no-project --exclude-newer PT12H --python 3.12 --with "pydantic>=2.10,<3" --with pyyaml python -m infx.klaud plan --directory "$RUNNER_TEMP/klaud" - name: Check for overlapping open PRs id: review @@ -53,7 +56,8 @@ jobs: GH_TOKEN: ${{ github.token }} KLAUD_EVIDENCE: ${{ runner.temp }}/klaud with: - anthropic_api_key: ${{ secrets.ANTHROPIC_API_KEY }} + # Repository automation credential; the planner runs only on main. + anthropic_api_key: ${{ secrets.ANTHROPIC_API_KEY }} # zizmor: ignore[secrets-outside-env] github_token: ${{ github.token }} track_progress: false settings: | @@ -103,19 +107,21 @@ jobs: - name: Select reviewed candidates within the total cap id: select env: - GH_TOKEN: ${{ secrets.CLAUDE_PAT }} - KLAUD_DASHBOARD_API_KEY: ${{ secrets.KLAUD_DASHBOARD_API_KEY }} + # Repository automation credential; the planner runs only on main. + GH_TOKEN: ${{ secrets.CLAUDE_PAT }} # zizmor: ignore[secrets-outside-env] + # Repository automation credential; the planner runs only on main. + KLAUD_DASHBOARD_API_KEY: ${{ secrets.KLAUD_DASHBOARD_API_KEY }} # zizmor: ignore[secrets-outside-env] KLAUD_PR_REVIEW: ${{ steps.review.outputs.structured_output }} KLAUD_REVIEW_OUTCOME: ${{ steps.review.outcome }} KLAUD_REVIEW_EXECUTION: ${{ steps.review.outputs.execution_file }} run: uv run --no-project --exclude-newer PT12H --python 3.12 --with "pydantic>=2.10,<3" python -m infx.klaud select --max-candidates-per-run "$MAX_CANDIDATES_PER_RUN" --directory "$RUNNER_TEMP/klaud" --execution-file "${KLAUD_REVIEW_EXECUTION:-$RUNNER_TEMP/claude-execution-output.json}" - - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a + - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: name: klaud-ownership path: ${{ runner.temp }}/klaud/ownership.json if-no-files-found: error retention-days: 90 - - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a + - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 if: always() with: name: klaud-plan @@ -136,7 +142,7 @@ jobs: fail-fast: false matrix: candidate: ${{ fromJSON(needs.plan.outputs.candidates) }} - uses: ./.github/workflows/klaud-candidate.yml + uses: $/.github/workflows/klaud-candidate.yml with: candidate: ${{ matrix.candidate }} secrets: diff --git a/.github/workflows/pr-recipe-reminder.yml b/.github/workflows/pr-recipe-reminder.yml index 35b00eb2d4..3945deea65 100644 --- a/.github/workflows/pr-recipe-reminder.yml +++ b/.github/workflows/pr-recipe-reminder.yml @@ -9,11 +9,17 @@ on: - 'benchmarks/**' - 'perf-changelog.yaml' -permissions: - pull-requests: write +permissions: {} + +concurrency: + group: recipe-reminder-${{ github.event.pull_request.number }} + cancel-in-progress: false jobs: comment: + name: comment + permissions: + pull-requests: write # Publish PR feedback. runs-on: ubuntu-latest steps: - name: Comment recipe reminder diff --git a/.github/workflows/profile.yml b/.github/workflows/profile.yml index 519e88eb8d..c543afe19c 100644 --- a/.github/workflows/profile.yml +++ b/.github/workflows/profile.yml @@ -1,6 +1,7 @@ name: Profile -on: +# Independent profiling requests use priority-scheduled runner slots. +on: # zizmor: ignore[concurrency-limits] workflow_dispatch: inputs: config-key: @@ -55,6 +56,7 @@ env: jobs: get-jobs: + name: get-jobs runs-on: ubuntu-latest outputs: filtered-matrix: ${{ steps.filter.outputs.filtered }} @@ -64,6 +66,7 @@ jobs: uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: ref: ${{ inputs.ref || github.sha }} + persist-credentials: false - name: Checkout priority scheduler tooling if: ${{ inputs.ref && inputs.ref != '' }} uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 @@ -76,26 +79,37 @@ jobs: - id: gen name: Generate matrix via script run: | - CLI_ARGS="test-config --config-files ${{ inputs.config-file }} --config-keys ${{ inputs.config-key }} --conc ${{ inputs.conc }}" - CONFIG_JSON=$(uv run --no-project --exclude-newer PT12H --python 3.12 --with pydantic --with pyyaml python "${GITHUB_WORKSPACE}/utils/matrix_logic/generate_sweep_configs.py" $CLI_ARGS) PRIORITY_ROOT="${GITHUB_WORKSPACE}" if [ -d "${GITHUB_WORKSPACE}/.ci-priority" ]; then PRIORITY_ROOT="${GITHUB_WORKSPACE}/.ci-priority" fi + source "$PRIORITY_ROOT/benchmarks/benchmark_lib.sh" --validation-only + check_env_vars INPUTS_CONFIG_FILE INPUTS_CONFIG_KEY INPUTS_CONC PR_LABELS + CONFIG_JSON=$(uv run --no-project --exclude-newer PT12H --python 3.12 --with pydantic --with pyyaml \ + python "${GITHUB_WORKSPACE}/utils/matrix_logic/generate_sweep_configs.py" \ + test-config --config-files "$INPUTS_CONFIG_FILE" --config-keys "$INPUTS_CONFIG_KEY" --conc "$INPUTS_CONC") CONFIG_JSON=$(printf '%s' "$CONFIG_JSON" | env PYTHONPATH="$PRIORITY_ROOT" uv run --no-project --exclude-newer PT12H --python 3.12 --with pyyaml \ python -P "${PRIORITY_ROOT}/utils/ci_priority.py" \ --policy "${PRIORITY_ROOT}/configs/ci-priority.yaml" \ - --event-name "${{ github.event_name }}" \ - --queue-namespace "${{ github.run_id }}:${{ github.run_attempt }}" \ - --labels-json '${{ inputs.moe-debug && '["ci-patchwork"]' || '[]' }}') - echo "raw=$CONFIG_JSON" >> $GITHUB_OUTPUT + --event-name "${GITHUB_EVENT_NAME}" \ + --queue-namespace "${GITHUB_RUN_ID}:${GITHUB_RUN_ATTEMPT}" \ + --labels-json "${PR_LABELS}") + echo "raw=$CONFIG_JSON" >> "$GITHUB_OUTPUT" + + env: + PR_LABELS: ${{ inputs.moe-debug && '["ci-patchwork"]' || '[]' }} + INPUTS_CONFIG_FILE: ${{ inputs.config-file }} + INPUTS_CONFIG_KEY: ${{ inputs.config-key }} + INPUTS_CONC: ${{ inputs.conc }} - id: filter name: Take first generated job shell: python + env: + STEPS_GEN_OUTPUTS_RAW: ${{ steps.gen.outputs.raw }} run: | import json, os, sys - raw = '${{ steps.gen.outputs.raw }}' + raw = os.environ["STEPS_GEN_OUTPUTS_RAW"] try: data = json.loads(raw) except Exception as e: @@ -134,8 +148,11 @@ jobs: - name: Fail if no matching entries if: ${{ steps.filter.outputs.count == '0' }} run: | - echo "No entries produced for config-key=${{ inputs.config-key }}, conc=${{ inputs.conc }}." >&2 + echo "No entries produced for config-key=${INPUTS_CONFIG_KEY}, conc=${INPUTS_CONC}." >&2 exit 1 + env: + INPUTS_CONFIG_KEY: ${{ inputs.config-key }} + INPUTS_CONC: ${{ inputs.conc }} profile: needs: get-jobs @@ -212,10 +229,10 @@ jobs: # Cleanup SLURM resources if command -v squeue >/dev/null 2>&1; then - echo "[Slurm] Cleaning up jobs with name: ${{ runner.name }} ..." - scancel --name="${{ runner.name }}" || true - while [ -n "$(squeue --name='${{ runner.name }}' --noheader --format='%i')" ]; do - squeue --name="${{ runner.name }}" + echo "[Slurm] Cleaning up jobs with name: ${RUNNER_NAME} ..." + scancel --name="${RUNNER_NAME}" || true + while [ -n "$(squeue --name="${RUNNER_NAME}" --noheader --format='%i')" ]; do + squeue --name="${RUNNER_NAME}" sleep 5 done fi @@ -226,6 +243,7 @@ jobs: fetch-depth: 0 ref: ${{ inputs.ref || github.sha }} clean: false + persist-credentials: false - name: Launch + Profile (single-node sglang/vllm) id: run @@ -328,19 +346,25 @@ jobs: repository: SemiAnalysisAI/InferenceX-trace-storage path: storage ref: master - ssh-key: ${{ secrets.PROFILER_STORAGE_DEPLOY_KEY }} + # Storage-repository deploy key for authorized profiling dispatches. + ssh-key: ${{ secrets.PROFILER_STORAGE_DEPLOY_KEY }} # zizmor: ignore[secrets-outside-env] fetch-depth: 0 + # This separate storage checkout pushes trace commits using its scoped deploy key. + persist-credentials: true # zizmor: ignore[artipacked] - name: Push profile to storage repo if: ${{ steps.run.outputs.trace != '' }} id: push env: + MATRIX_CONFIG_RUNNER: ${{ matrix.config.runner }} + MATRIX_CONFIG_EXP_NAME: ${{ matrix.config['exp-name'] }} + MATRIX_EP: ${{ matrix.config.ep || 1 }} TRACE_LOCAL: ${{ steps.run.outputs.trace }} shell: bash run: | set -eo pipefail - dest_dir="storage/profiles/${GITHUB_SHA}/${{ matrix.config.runner }}/${{ matrix.config.framework }}/${{ matrix.config['exp-name'] }}_${{ matrix.config.precision }}_tp${{ matrix.config.tp }}_pp${{ matrix.config.pp }}_dcp${{ matrix.config.dcp-size }}_pcp${{ matrix.config.pcp-size }}_ep${{ matrix.config.ep || 1 }}_conc${{ matrix.config.conc }}" + dest_dir="storage/profiles/${GITHUB_SHA}/${MATRIX_CONFIG_RUNNER}/${FRAMEWORK}/${MATRIX_CONFIG_EXP_NAME}_${PRECISION}_tp${TP}_pp${PP_SIZE}_dcp${DCP_SIZE}_pcp${PCP_SIZE}_ep${MATRIX_EP}_conc${CONC}" mkdir -p "$dest_dir" cp "$TRACE_LOCAL" "$dest_dir/trace.json.gz" @@ -348,13 +372,13 @@ jobs: git config user.name "github-actions" git config user.email "github-actions@github.com" git add -A - git commit -m "Add profile: ${GITHUB_SHA} ${{ matrix.config['exp-name'] }} tp${{ matrix.config.tp }} pp${{ matrix.config.pp }} dcp${{ matrix.config.dcp-size }} pcp${{ matrix.config.pcp-size }} ep${{ matrix.config.ep || 1 }} conc${{ matrix.config.conc }}" || echo "Nothing to commit" + git commit -m "Add profile: ${GITHUB_SHA} ${MATRIX_CONFIG_EXP_NAME} tp${TP} pp${PP_SIZE} dcp${DCP_SIZE} pcp${PCP_SIZE} ep${MATRIX_EP} conc${CONC}" || echo "Nothing to commit" git push STORAGE_SHA="$(git rev-parse HEAD)" popd >/dev/null - export RAW_URL="https://raw.githubusercontent.com/SemiAnalysisAI/InferenceX-trace-storage/${STORAGE_SHA}/profiles/${GITHUB_SHA}/${{ matrix.config.runner }}/${{ matrix.config.framework }}/${{ matrix.config['exp-name'] }}_${{ matrix.config.precision }}_tp${{ matrix.config.tp }}_pp${{ matrix.config.pp }}_dcp${{ matrix.config.dcp-size }}_pcp${{ matrix.config.pcp-size }}_ep${{ matrix.config.ep || 1 }}_conc${{ matrix.config.conc }}/trace.json.gz" - export TITLE="${{ matrix.config['exp-name'] }}_${{ matrix.config.precision }}_tp${{ matrix.config.tp }}_pp${{ matrix.config.pp }}_dcp${{ matrix.config.dcp-size }}_pcp${{ matrix.config.pcp-size }}_ep${{ matrix.config.ep || 1 }}_conc${{ matrix.config.conc }}" + export RAW_URL="https://raw.githubusercontent.com/SemiAnalysisAI/InferenceX-trace-storage/${STORAGE_SHA}/profiles/${GITHUB_SHA}/${MATRIX_CONFIG_RUNNER}/${FRAMEWORK}/${MATRIX_CONFIG_EXP_NAME}_${PRECISION}_tp${TP}_pp${PP_SIZE}_dcp${DCP_SIZE}_pcp${PCP_SIZE}_ep${MATRIX_EP}_conc${CONC}/trace.json.gz" + export TITLE="${MATRIX_CONFIG_EXP_NAME}_${PRECISION}_tp${TP}_pp${PP_SIZE}_dcp${DCP_SIZE}_pcp${PCP_SIZE}_ep${MATRIX_EP}_conc${CONC}" enc_src="$(python3 -c 'import os,urllib.parse; print(urllib.parse.quote(os.environ["RAW_URL"], safe=""))')" enc_title="$(python3 -c 'import os,urllib.parse; print(urllib.parse.quote(os.environ["TITLE"], safe=""))')" diff --git a/.github/workflows/recover-reused-ingest.yml b/.github/workflows/recover-reused-ingest.yml index 2bcdba781a..9cc6cd804d 100644 --- a/.github/workflows/recover-reused-ingest.yml +++ b/.github/workflows/recover-reused-ingest.yml @@ -12,21 +12,33 @@ on: required: true type: string +permissions: {} + +concurrency: + group: recover-ingest-${{ inputs.source-run-id }}-${{ inputs.merge-run-id }} + cancel-in-progress: false + jobs: trigger-agentic-ingest: + name: trigger-agentic-ingest runs-on: ubuntu-latest steps: - name: Trigger agentic database ingest - run: | - curl -sSf -X POST \ - -H "Authorization: Bearer ${{ secrets.INFX_FRONTEND_PAT }}" \ - -H "Accept: application/vnd.github+v3+json" \ - https://api.github.com/repos/SemiAnalysisAI/InferenceX-app/dispatches \ - -d '{ - "event_type": "ingest-agentic-results", - "client_payload": { - "source-run-id": "${{ inputs.source-run-id }}", - "merge-run-id": "${{ inputs.merge-run-id }}", - "database-target": "production" - } - }' + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + env: + SOURCE_RUN_ID: ${{ inputs.source-run-id }} + MERGE_RUN_ID: ${{ inputs.merge-run-id }} + with: + # Cross-repository dispatch credential for maintainer-triggered ingest recovery. + github-token: ${{ secrets.INFX_FRONTEND_PAT }} # zizmor: ignore[secrets-outside-env] + script: | + await github.rest.repos.createDispatchEvent({ + owner: "SemiAnalysisAI", + repo: "InferenceX-app", + event_type: "ingest-agentic-results", + client_payload: { + "source-run-id": process.env.SOURCE_RUN_ID, + "merge-run-id": process.env.MERGE_RUN_ID, + "database-target": "production" + } + }); diff --git a/.github/workflows/reuse-sweep-comment.yml b/.github/workflows/reuse-sweep-comment.yml index 61792d80dc..3b8ed45831 100644 --- a/.github/workflows/reuse-sweep-comment.yml +++ b/.github/workflows/reuse-sweep-comment.yml @@ -13,6 +13,7 @@ concurrency: jobs: acknowledge: + name: acknowledge if: >- github.event.issue.pull_request && (contains(github.event.comment.body, '/reuse-sweep-run') || @@ -20,10 +21,10 @@ jobs: runs-on: ubuntu-latest timeout-minutes: 5 permissions: - actions: read + actions: read # Read workflow runs and artifacts. contents: read - issues: write - pull-requests: write + issues: write # Update comments, labels, and reactions. + pull-requests: write # Publish PR feedback. steps: - name: Checkout trusted workflow code uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 diff --git a/.github/workflows/run-sweep.yml b/.github/workflows/run-sweep.yml index 8481c6e1a9..0a839b908c 100644 --- a/.github/workflows/run-sweep.yml +++ b/.github/workflows/run-sweep.yml @@ -40,8 +40,12 @@ on: paths: - "perf-changelog.yaml" +permissions: + contents: read + jobs: check-changelog: + name: check-changelog runs-on: ubuntu-latest permissions: contents: read @@ -112,6 +116,7 @@ jobs: - name: Validate perf-changelog matrix env: + PR_HEAD_SHA: ${{ github.event.pull_request.head.sha }} ALL_EVALS: ${{ contains(github.event.pull_request.labels.*.name, 'all-evals') }} EVALS_ONLY: ${{ contains(github.event.pull_request.labels.*.name, 'evals-only') }} run: | @@ -119,8 +124,8 @@ jobs: uv run --no-project --exclude-newer PT12H --python 3.12 --with "pydantic>=2" --with pyyaml python -m infx.workflows.validate_perf_changelog --changelog-file perf-changelog.yaml - --base-ref "origin/${{ github.base_ref }}" - --head-ref "${{ github.event.pull_request.head.sha }}" + --base-ref "origin/${GITHUB_BASE_REF}" + --head-ref "${PR_HEAD_SHA}" ) if [ "$ALL_EVALS" = "true" ]; then CMD+=(--all-evals) @@ -131,13 +136,14 @@ jobs: "${CMD[@]}" reuse-sweep-gate: + name: reuse-sweep-gate needs: check-changelog runs-on: ubuntu-latest permissions: - actions: read + actions: read # Read workflow runs and artifacts. contents: read - issues: read - pull-requests: read + issues: read # Read issue comments. + pull-requests: read # Read PR metadata. if: >- always() && needs.check-changelog.result == 'success' && @@ -150,29 +156,35 @@ jobs: steps: - name: Checkout code uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false - name: Check for reusable sweep authorization id: gate env: + PR_HEAD_SHA: ${{ github.event.pull_request.head.sha }} + PR_NUMBER: ${{ github.event.pull_request.number }} GH_TOKEN: ${{ github.token }} + EVENT_ACTION: ${{ github.event.action }} run: | python3 -m infx.workflows.reuse \ - --repo "${{ github.repository }}" \ - --commit-sha "${{ github.event.pull_request.head.sha }}" \ - --event-name "${{ github.event_name }}" \ - --event-action "${{ github.event.action }}" \ - --pr-number "${{ github.event.pull_request.number }}" \ - --ref "${{ github.ref }}" \ + --repo "${GITHUB_REPOSITORY}" \ + --commit-sha "${PR_HEAD_SHA}" \ + --event-name "${GITHUB_EVENT_NAME}" \ + --event-action "${EVENT_ACTION}" \ + --pr-number "${PR_NUMBER}" \ + --ref "${GITHUB_REF}" \ --workflow-id "run-sweep.yml" setup: + name: setup needs: [check-changelog, reuse-sweep-gate] runs-on: ubuntu-latest permissions: - actions: read + actions: read # Read workflow runs and artifacts. contents: read - issues: read - pull-requests: read + issues: read # Read issue comments. + pull-requests: read # Read PR metadata. if: >- always() && ( @@ -233,6 +245,7 @@ jobs: uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: fetch-depth: 0 + persist-credentials: false - uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1 - name: Classify priority criteria @@ -244,7 +257,8 @@ jobs: uses: anthropics/claude-code-action@0d0e0876d3eaa933f45dc692f7a4312c83caf36f # v1.0.218 with: github_token: ${{ github.token }} - anthropic_api_key: ${{ secrets.ANTHROPIC_API_KEY }} + # Repository integration credential for the scoped sweep/ingest job. + anthropic_api_key: ${{ secrets.ANTHROPIC_API_KEY }} # zizmor: ignore[secrets-outside-env] track_progress: false settings: | {"fastMode": true} @@ -291,6 +305,10 @@ jobs: - id: setup env: + PR_HEAD_SHA: ${{ github.event.pull_request.head.sha }} + PUSH_BEFORE: ${{ github.event.before }} + PUSH_AFTER: ${{ github.event.after }} + PR_NUMBER: ${{ github.event.pull_request.number || 0 }} GH_TOKEN: ${{ github.token }} PR_LABELS: ${{ toJson(github.event.pull_request.labels.*.name) }} PRIORITY_CRITERIA: ${{ steps.priority-criteria.outputs.criteria || '' }} @@ -311,12 +329,12 @@ jobs: }} run: | - if [ "${{ github.event_name }}" == "pull_request" ]; then - BASE_REF="origin/${{ github.base_ref }}" - HEAD_REF="${{ github.event.pull_request.head.sha }}" + if [ "${GITHUB_EVENT_NAME}" == "pull_request" ]; then + BASE_REF="origin/${GITHUB_BASE_REF}" + HEAD_REF="${PR_HEAD_SHA}" else - BASE_REF="${{ github.event.before }}" - HEAD_REF="${{ github.event.after }}" + BASE_REF="${PUSH_BEFORE}" + HEAD_REF="${PUSH_AFTER}" fi CMD=( @@ -341,21 +359,22 @@ jobs: python -m infx.workflows.benchmark_schema --plan) CONFIG_JSON=$(printf '%s' "$CONFIG_JSON" | uv run --no-project --exclude-newer PT12H --python 3.12 --with pyyaml \ python -m infx.workflows.ci_priority \ - --event-name "${{ github.event_name }}" \ - --queue-namespace "${{ github.run_id }}:${{ github.run_attempt }}" \ + --event-name "${GITHUB_EVENT_NAME}" \ + --queue-namespace "${GITHUB_RUN_ID}:${GITHUB_RUN_ATTEMPT}" \ --labels-json "$PR_LABELS" \ - --pr-number "${{ github.event.pull_request.number || 0 }}" \ + --pr-number "${PR_NUMBER}" \ --criteria-json "$PRIORITY_CRITERIA") echo "search-space-config=$CONFIG_JSON" >> "$GITHUB_OUTPUT" python3 -m infx.workflows.reuse \ - --repo "${{ github.repository }}" \ - --commit-sha "${{ github.sha }}" \ - --event-name "${{ github.event_name }}" \ - --ref "${{ github.ref }}" \ + --repo "${GITHUB_REPOSITORY}" \ + --commit-sha "${GITHUB_SHA}" \ + --event-name "${GITHUB_EVENT_NAME}" \ + --ref "${GITHUB_REF}" \ --workflow-id "run-sweep.yml" canary-select: + name: canary-select needs: setup if: >- needs.setup.outputs.reuse-enabled != 'true' && @@ -403,13 +422,17 @@ jobs: canary-sweep: needs: canary-select if: ${{ needs.canary-select.outputs.canary-config != '' && needs.canary-select.outputs.canary-config != '[]' }} - uses: ./.github/workflows/benchmark-tmpl.yml + uses: $/.github/workflows/benchmark-tmpl.yml name: canary / strategy: fail-fast: false matrix: config: ${{ fromJson(needs.canary-select.outputs.canary-config) }} - secrets: inherit + secrets: + REPO_PAT: ${{ secrets.REPO_PAT }} + INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} + MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} + MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} # Only same-repository PRs owned by Klaud receive background priority. @@ -432,13 +455,17 @@ jobs: (needs.canary-sweep.result == 'success' || needs.canary-sweep.result == 'skipped') && toJson(fromJson(needs.setup.outputs.search-space-config).multi_node['1k1k']) != 'null' }} - uses: ./.github/workflows/benchmark-multinode-tmpl.yml + uses: $/.github/workflows/benchmark-multinode-tmpl.yml name: multi-node 1k1k / strategy: fail-fast: ${{ contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') }} matrix: config: ${{ fromJson(needs.setup.outputs.search-space-config).multi_node['1k1k'] }} - secrets: inherit + secrets: + REPO_PAT: ${{ secrets.REPO_PAT }} + INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} + MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} + MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: &multi-node-inputs config: ${{ toJSON(matrix.config) }} klaud-run: *klaud-run @@ -461,13 +488,17 @@ jobs: (needs.canary-sweep.result == 'success' || needs.canary-sweep.result == 'skipped') && toJson(fromJson(needs.setup.outputs.search-space-config).multi_node['8k1k']) != 'null' }} - uses: ./.github/workflows/benchmark-multinode-tmpl.yml + uses: $/.github/workflows/benchmark-multinode-tmpl.yml name: multi-node 8k1k / strategy: fail-fast: ${{ contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') }} matrix: config: ${{ fromJson(needs.setup.outputs.search-space-config).multi_node['8k1k'] }} - secrets: inherit + secrets: + REPO_PAT: ${{ secrets.REPO_PAT }} + INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} + MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} + MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: *multi-node-inputs sweep-single-node-1k1k: @@ -481,13 +512,17 @@ jobs: toJson(fromJson((needs.canary-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).single_node['1k1k']) != 'null' && toJson(fromJson((needs.canary-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).single_node['1k1k']) != '[]' }} - uses: ./.github/workflows/benchmark-tmpl.yml + uses: $/.github/workflows/benchmark-tmpl.yml name: single-node 1k1k / strategy: fail-fast: ${{ contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') }} matrix: config: ${{ fromJson((needs.canary-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).single_node['1k1k'] }} - secrets: inherit + secrets: + REPO_PAT: ${{ secrets.REPO_PAT }} + INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} + MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} + MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: &single-node-inputs config: ${{ toJSON(matrix.config) }} klaud-run: *klaud-run @@ -510,13 +545,17 @@ jobs: toJson(fromJson((needs.canary-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).single_node['8k1k']) != 'null' && toJson(fromJson((needs.canary-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).single_node['8k1k']) != '[]' }} - uses: ./.github/workflows/benchmark-tmpl.yml + uses: $/.github/workflows/benchmark-tmpl.yml name: single-node 8k1k / strategy: fail-fast: ${{ contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') }} matrix: config: ${{ fromJson((needs.canary-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).single_node['8k1k'] }} - secrets: inherit + secrets: + REPO_PAT: ${{ secrets.REPO_PAT }} + INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} + MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} + MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: *single-node-inputs sweep-agentic: @@ -529,13 +568,17 @@ jobs: (needs.canary-sweep.result == 'success' || needs.canary-sweep.result == 'skipped') && toJson(fromJson(needs.setup.outputs.search-space-config).single_node['agentic']) != 'null' }} - uses: ./.github/workflows/benchmark-tmpl.yml + uses: $/.github/workflows/benchmark-tmpl.yml name: agentic / strategy: fail-fast: ${{ contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') }} matrix: config: ${{ fromJson(needs.setup.outputs.search-space-config).single_node['agentic'] }} - secrets: inherit + secrets: + REPO_PAT: ${{ secrets.REPO_PAT }} + INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} + MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} + MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} klaud-run: *klaud-run @@ -559,13 +602,17 @@ jobs: (needs.canary-sweep.result == 'success' || needs.canary-sweep.result == 'skipped') && toJson(fromJson(needs.setup.outputs.search-space-config).multi_node['agentic']) != 'null' }} - uses: ./.github/workflows/benchmark-multinode-tmpl.yml + uses: $/.github/workflows/benchmark-multinode-tmpl.yml name: multi-node agentic / strategy: fail-fast: ${{ contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') }} matrix: config: ${{ fromJson(needs.setup.outputs.search-space-config).multi_node['agentic'] }} - secrets: inherit + secrets: + REPO_PAT: ${{ secrets.REPO_PAT }} + INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} + MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} + MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} klaud-run: *klaud-run @@ -593,13 +640,17 @@ jobs: toJson(fromJson(needs.setup.outputs.search-space-config).evals) != '[]' && toJson(fromJson(needs.setup.outputs.search-space-config).evals) != 'null' }} - uses: ./.github/workflows/benchmark-tmpl.yml + uses: $/.github/workflows/benchmark-tmpl.yml name: eval / strategy: fail-fast: ${{ contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') }} matrix: config: ${{ fromJson(needs.setup.outputs.search-space-config).evals }} - secrets: inherit + secrets: + REPO_PAT: ${{ secrets.REPO_PAT }} + INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} + MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} + MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} klaud-run: *klaud-run @@ -627,13 +678,17 @@ jobs: toJson(fromJson(needs.setup.outputs.search-space-config).agentic_evals) != '[]' && toJson(fromJson(needs.setup.outputs.search-space-config).agentic_evals) != 'null' }} - uses: ./.github/workflows/benchmark-tmpl.yml + uses: $/.github/workflows/benchmark-tmpl.yml name: agentic eval / strategy: fail-fast: ${{ contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') }} matrix: config: ${{ fromJson(needs.setup.outputs.search-space-config).agentic_evals }} - secrets: inherit + secrets: + REPO_PAT: ${{ secrets.REPO_PAT }} + INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} + MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} + MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} klaud-run: *klaud-run @@ -660,13 +715,17 @@ jobs: toJson(fromJson(needs.setup.outputs.search-space-config).multinode_evals) != '[]' && toJson(fromJson(needs.setup.outputs.search-space-config).multinode_evals) != 'null' }} - uses: ./.github/workflows/benchmark-multinode-tmpl.yml + uses: $/.github/workflows/benchmark-multinode-tmpl.yml name: multi-node eval / strategy: fail-fast: ${{ contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') }} matrix: config: ${{ fromJson(needs.setup.outputs.search-space-config).multinode_evals }} - secrets: inherit + secrets: + REPO_PAT: ${{ secrets.REPO_PAT }} + INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} + MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} + MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} klaud-run: *klaud-run @@ -699,13 +758,17 @@ jobs: toJson(fromJson(needs.setup.outputs.search-space-config).multinode_agentic_evals) != '[]' && toJson(fromJson(needs.setup.outputs.search-space-config).multinode_agentic_evals) != 'null' }} - uses: ./.github/workflows/benchmark-multinode-tmpl.yml + uses: $/.github/workflows/benchmark-multinode-tmpl.yml name: multi-node agentic eval / strategy: fail-fast: ${{ contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') }} matrix: config: ${{ fromJson(needs.setup.outputs.search-space-config).multinode_agentic_evals }} - secrets: inherit + secrets: + REPO_PAT: ${{ secrets.REPO_PAT }} + INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} + MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} + MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} klaud-run: *klaud-run @@ -726,6 +789,7 @@ jobs: scenario-type: agentic-coding collect-results: + name: collect-results needs: [ canary-sweep, @@ -749,18 +813,18 @@ jobs: needs.sweep-multi-node-8k1k.result != 'skipped' ) }} - uses: ./.github/workflows/collect-results.yml - secrets: inherit + uses: $/.github/workflows/collect-results.yml with: result-prefix: "bmk" collect-evals: + name: collect-evals needs: [sweep-evals, sweep-agentic-evals, sweep-multi-node-evals, sweep-multi-node-agentic-evals, setup] if: ${{ always() && needs.setup.result != 'skipped' && (needs.sweep-evals.result != 'skipped' || needs.sweep-agentic-evals.result != 'skipped' || needs.sweep-multi-node-evals.result != 'skipped' || needs.sweep-multi-node-agentic-evals.result != 'skipped') }} - uses: ./.github/workflows/collect-evals.yml - secrets: inherit + uses: $/.github/workflows/collect-evals.yml upload-changelog-metadata: + name: upload-changelog-metadata needs: [setup, collect-results] if: ${{ always() && needs.setup.result == 'success' }} runs-on: ubuntu-latest @@ -803,6 +867,7 @@ jobs: if-no-files-found: error calc-success-rate: + name: calc-success-rate needs: collect-results if: ${{ always() && needs.collect-results.result != 'skipped'}} runs-on: ubuntu-latest @@ -810,13 +875,16 @@ jobs: env: RESULTS_DIR: "results/" STATS_FILENAME: "run_stats" - GITHUB_TOKEN: ${{ secrets.REPO_PAT }} + # Repository integration credential for the scoped sweep/ingest job. + GITHUB_TOKEN: ${{ secrets.REPO_PAT }} # zizmor: ignore[secrets-outside-env] steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: - token: ${{ secrets.REPO_PAT }} + # Repository integration credential for the scoped sweep/ingest job. + token: ${{ secrets.REPO_PAT }} # zizmor: ignore[secrets-outside-env] fetch-depth: 0 + persist-credentials: false - name: Download results artifacts uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 @@ -838,6 +906,7 @@ jobs: path: ${{ env.STATS_FILENAME }}.json compare-results: + name: compare-results needs: [ collect-results, @@ -850,10 +919,13 @@ jobs: runs-on: ubuntu-latest env: - DATABASE_URL: ${{ secrets.NEON_PROD_RO_URL }} + # Repository integration credential for the scoped sweep/ingest job. + DATABASE_URL: ${{ secrets.NEON_PROD_RO_URL }} # zizmor: ignore[secrets-outside-env] steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false - name: Download results artifacts uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 @@ -870,6 +942,7 @@ jobs: python -m infx.results.compare_results results/ >> "$GITHUB_STEP_SUMMARY" trigger-ingest: + name: trigger-ingest needs: [ collect-results, @@ -901,20 +974,26 @@ jobs: runs-on: ubuntu-latest steps: - name: Trigger database ingest - run: | - curl -sSf -X POST \ - -H "Authorization: Bearer ${{ secrets.INFX_FRONTEND_PAT }}" \ - -H "Accept: application/vnd.github+v3+json" \ - https://api.github.com/repos/SemiAnalysisAI/InferenceX-app/dispatches \ - -d '{ - "event_type": "ingest-results", - "client_payload": { - "source-run-id": "${{ needs.setup.outputs.reuse-enabled == 'true' && needs.setup.outputs.reuse-source-run-id || github.run_id }}", - "merge-run-id": "${{ github.run_id }}" - } - }' + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + env: + SOURCE_RUN_ID: ${{ needs.setup.outputs.reuse-enabled == 'true' && needs.setup.outputs.reuse-source-run-id || github.run_id }} + MERGE_RUN_ID: ${{ github.run_id }} + with: + # Repository integration credential for the scoped sweep/ingest job. + github-token: ${{ secrets.INFX_FRONTEND_PAT }} # zizmor: ignore[secrets-outside-env] + script: | + await github.rest.repos.createDispatchEvent({ + owner: "SemiAnalysisAI", + repo: "InferenceX-app", + event_type: "ingest-results", + client_payload: { + "source-run-id": process.env.SOURCE_RUN_ID, + "merge-run-id": process.env.MERGE_RUN_ID + } + }); trigger-agentic-ingest: + name: trigger-agentic-ingest needs: [ setup, @@ -960,21 +1039,27 @@ jobs: runs-on: ubuntu-latest steps: - name: Trigger agentic database ingest - run: | - curl -sSf -X POST \ - -H "Authorization: Bearer ${{ secrets.INFX_FRONTEND_PAT }}" \ - -H "Accept: application/vnd.github+v3+json" \ - https://api.github.com/repos/SemiAnalysisAI/InferenceX-app/dispatches \ - -d '{ - "event_type": "ingest-agentic-results", - "client_payload": { - "source-run-id": "${{ needs.setup.outputs.reuse-enabled == 'true' && needs.setup.outputs.reuse-source-run-id || github.run_id }}", - "merge-run-id": "${{ github.run_id }}", - "database-target": "production" - } - }' + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + env: + SOURCE_RUN_ID: ${{ needs.setup.outputs.reuse-enabled == 'true' && needs.setup.outputs.reuse-source-run-id || github.run_id }} + MERGE_RUN_ID: ${{ github.run_id }} + with: + # Repository integration credential for the scoped sweep/ingest job. + github-token: ${{ secrets.INFX_FRONTEND_PAT }} # zizmor: ignore[secrets-outside-env] + script: | + await github.rest.repos.createDispatchEvent({ + owner: "SemiAnalysisAI", + repo: "InferenceX-app", + event_type: "ingest-agentic-results", + client_payload: { + "source-run-id": process.env.SOURCE_RUN_ID, + "merge-run-id": process.env.MERGE_RUN_ID, + "database-target": "production" + } + }); comment-unofficial-run-visualizer: + name: comment-unofficial-run-visualizer needs: [ collect-results, @@ -1010,7 +1095,7 @@ jobs: group: unofficial-run-visualizer-${{ github.event.pull_request.number }} cancel-in-progress: false permissions: - pull-requests: write + pull-requests: write # Publish PR feedback. steps: - name: Update unofficial run visualizer links on PR uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 diff --git a/.github/workflows/runner-offline-digest.yml b/.github/workflows/runner-offline-digest.yml index 390628b1bd..13563c6a68 100644 --- a/.github/workflows/runner-offline-digest.yml +++ b/.github/workflows/runner-offline-digest.yml @@ -18,14 +18,20 @@ on: permissions: contents: read +concurrency: + group: runner-offline-digest + cancel-in-progress: false + jobs: digest: + name: digest runs-on: ubuntu-latest steps: - name: List offline runners id: runners env: - GH_TOKEN: ${{ secrets.RUNNERS_PAT }} + # Repository runner/Slack integration credential for the scheduled digest. + GH_TOKEN: ${{ secrets.RUNNERS_PAT }} # zizmor: ignore[secrets-outside-env] REPO: ${{ github.repository }} run: | set -eo pipefail @@ -59,7 +65,8 @@ jobs: - name: Post Slack digest if: steps.runners.outputs.offline_count != '0' env: - SLACK_BOT_TOKEN: ${{ secrets.SLACK_BOT_TOKEN }} + # Repository runner/Slack integration credential for the scheduled digest. + SLACK_BOT_TOKEN: ${{ secrets.SLACK_BOT_TOKEN }} # zizmor: ignore[secrets-outside-env] CHANNEL: C09PULGMVNG OFFLINE_COUNT: ${{ steps.runners.outputs.offline_count }} TOTAL: ${{ steps.runners.outputs.total }} diff --git a/.github/workflows/speedbench-al.yml b/.github/workflows/speedbench-al.yml index ed3f71c7f4..afad1a6838 100644 --- a/.github/workflows/speedbench-al.yml +++ b/.github/workflows/speedbench-al.yml @@ -6,7 +6,8 @@ name: SpeedBench AL Collection # synthetic-acceptance framework and (optionally) opens a PR updating # benchmarks/speedbench-reference-al.yaml. -on: +# Independent collection requests use priority-scheduled runner slots. +on: # zizmor: ignore[concurrency-limits] workflow_dispatch: inputs: runner: @@ -119,6 +120,7 @@ env: jobs: setup: + name: setup runs-on: ubuntu-latest outputs: priority: ${{ steps.score.outputs.priority }} @@ -145,7 +147,7 @@ jobs: scored=$(printf '%s' "$entry" | uv run --no-project --exclude-newer PT12H --python 3.12 --with pyyaml \ python -m infx.workflows.ci_priority \ - --queue-namespace "${{ github.run_id }}:${{ github.run_attempt }}") + --queue-namespace "${GITHUB_RUN_ID}:${GITHUB_RUN_ATTEMPT}") echo "priority=$(jq -r '.[0].priority' <<<"$scored")" >> "$GITHUB_OUTPUT" echo "queue-token=$(jq -r '.[0]["queue-token"]' <<<"$scored")" >> "$GITHUB_OUTPUT" @@ -159,7 +161,7 @@ jobs: format( '["self-hosted",{0},{1},{2},{3}]', toJSON(inputs.runner), - toJSON('nodes:1'), + '"nodes:1"', toJSON(format( 'ci-job-{0}-{1}', needs.setup.outputs.priority, @@ -198,25 +200,27 @@ jobs: # Cleanup SLURM resources if command -v squeue >/dev/null 2>&1; then - echo "[Slurm] Cleaning up jobs with name: ${{ runner.name }} ..." - scancel --name="${{ runner.name }}" || true - while [ -n "$(squeue --name='${{ runner.name }}' --noheader --format='%i')" ]; do - squeue --name="${{ runner.name }}" + echo "[Slurm] Cleaning up jobs with name: ${RUNNER_NAME} ..." + scancel --name="${RUNNER_NAME}" || true + while [ -n "$(squeue --name="${RUNNER_NAME}" --noheader --format='%i')" ]; do + squeue --name="${RUNNER_NAME}" sleep 5 done fi # Cleanup AL-matrix outputs from a prior job on this runner so a stale # matrix from a previous run is never picked up as this job's output. - rm -rf "${{ github.workspace }}/speedbench_results" 2>/dev/null || true + rm -rf "${GITHUB_WORKSPACE}/speedbench_results" 2>/dev/null || true - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: - token: ${{ secrets.REPO_PAT }} + # Repository write credential for authorized reference-data collection. + token: ${{ secrets.REPO_PAT }} # zizmor: ignore[secrets-outside-env] fetch-depth: 0 ref: ${{ inputs.ref || github.sha }} clean: true submodules: true + persist-credentials: false - name: Cleanup stale outputs (pre-run) run: | @@ -259,7 +263,8 @@ jobs: - name: Open PR updating reference yaml if: ${{ inputs.open-pr && success() }} env: - GH_TOKEN: ${{ secrets.REPO_PAT }} + # Repository write credential for authorized reference-data collection. + GH_TOKEN: ${{ secrets.REPO_PAT }} # zizmor: ignore[secrets-outside-env] run: | set -eo pipefail # NOTE: the reference yaml is keyed by model at the top level. This @@ -267,7 +272,7 @@ jobs: # model is collected, replace this cp with a per-model-key YAML merge. cp speedbench-reference-al.yaml benchmarks/speedbench-reference-al.yaml - BRANCH="speedbench-al/${{ inputs.model-prefix }}-auto-${{ github.run_id }}" + BRANCH="speedbench-al/${MODEL_PREFIX}-auto-${GITHUB_RUN_ID}" git config user.name "github-actions" git config user.email "github-actions@github.com" git checkout -b "$BRANCH" @@ -276,11 +281,11 @@ jobs: echo "No change in reference yaml; skipping PR." exit 0 fi - git commit -m "Update SpeedBench AL reference matrix for ${{ inputs.model }} (auto, run ${{ github.run_id }})" - git push -u origin "$BRANCH" + git commit -m "Update SpeedBench AL reference matrix for ${MODEL} (auto, run ${GITHUB_RUN_ID})" + git -c credential.helper= -c 'credential.helper=!gh auth git-credential' push -u origin "$BRANCH" gh pr create \ - --title "Update SpeedBench AL reference matrix for ${{ inputs.model-prefix }} (auto)" \ - --body "Auto-generated by the SpeedBench AL Collection workflow (run ${{ github.run_id }}). Model: \`${{ inputs.model }}\`, category: \`${{ inputs.category }}\`, MTP: \`${{ inputs.mtp-list }}\`, thinking: \`${{ inputs.thinking-modes }}\`, output_len: \`${{ inputs.output-len }}\`. Please review the measured values before merging." \ + --title "Update SpeedBench AL reference matrix for ${MODEL_PREFIX} (auto)" \ + --body "Auto-generated by the SpeedBench AL Collection workflow (run ${GITHUB_RUN_ID}). Model: \`${MODEL}\`, category: \`${CATEGORY}\`, MTP: \`${MTP_LIST}\`, thinking: \`${THINKING_MODES}\`, output_len: \`${SPEEDBENCH_OUTPUT_LEN}\`. Please review the measured values before merging." \ --base main \ --head "$BRANCH" diff --git a/.github/workflows/stage-results.yml b/.github/workflows/stage-results.yml index 8dd1e5ce80..459cc413fe 100644 --- a/.github/workflows/stage-results.yml +++ b/.github/workflows/stage-results.yml @@ -282,7 +282,8 @@ jobs: REQUESTED_BY: ${{ steps.request.outputs.requested-by }} COMMENT_ID: ${{ steps.acknowledge.outputs.comment-id }} with: - github-token: ${{ secrets.INFX_FRONTEND_PAT }} + # Cross-repository dispatch credential; trusted control code validates maintainer authorization. + github-token: ${{ secrets.INFX_FRONTEND_PAT }} # zizmor: ignore[secrets-outside-env] script: | await github.rest.repos.createDispatchEvent({ owner: 'SemiAnalysisAI', diff --git a/.github/workflows/trusted-external-sweep.yml b/.github/workflows/trusted-external-sweep.yml index 068b5c28fa..62c65af333 100644 --- a/.github/workflows/trusted-external-sweep.yml +++ b/.github/workflows/trusted-external-sweep.yml @@ -5,20 +5,24 @@ run-name: Trusted external sweep - PR #${{ github.event.pull_request.number }} @ # executes pull-request code. A write-authorized collaborator applies one primary # sweep label to approve the exact external head SHA, then this trusted event # dispatches the existing e2e workflow from main with repository secrets. -on: - pull_request_target: +# Each authorized label event approves an exact PR head; no event may cancel another. +on: # zizmor: ignore[concurrency-limits] + # Control plane only: verifies writer permission and the exact labeled head, never executes PR code. + pull_request_target: # zizmor: ignore[dangerous-triggers] branches: - main types: - labeled -permissions: - actions: write - contents: read - pull-requests: write +permissions: {} jobs: dispatch: + name: dispatch + permissions: + actions: write # Dispatch authorized sweeps. + contents: read + pull-requests: write # Publish PR feedback. if: >- github.event.pull_request.head.repo.full_name != github.repository && contains( diff --git a/.github/workflows/zizmor.yml b/.github/workflows/zizmor.yml new file mode 100644 index 0000000000..2ad99b783d --- /dev/null +++ b/.github/workflows/zizmor.yml @@ -0,0 +1,45 @@ +name: Workflow security + +on: + pull_request: + types: [opened, synchronize, reopened, ready_for_review] + paths: &workflow-paths + - '.github/workflows/*.yml' + - '.github/workflows/*.yaml' + - '.github/dependabot.yml' + - '.github/dependabot.yaml' + - '**/action.yml' + - '**/action.yaml' + - '**/.pre-commit-*.yml' + - '**/.pre-commit-*.yaml' + - '**/zizmor.yml' + - '**/zizmor.yaml' + push: + branches: [main] + paths: *workflow-paths + workflow_dispatch: + +permissions: + contents: read + +concurrency: + group: zizmor-${{ github.event.pull_request.number || github.ref }} + cancel-in-progress: true + +jobs: + audit: + name: Zizmor + runs-on: ubuntu-latest + timeout-minutes: 5 + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + - uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1 + - name: Audit GitHub Actions + env: + GH_TOKEN: ${{ github.token }} + run: >- + uvx --exclude-newer PT12H zizmor@latest + --persona auditor --strict-collection --collect all --no-config + --format github --no-progress . diff --git a/docs/ci-procedures.md b/docs/ci-procedures.md index b456fd7da8..faf42ac244 100644 --- a/docs/ci-procedures.md +++ b/docs/ci-procedures.md @@ -281,7 +281,7 @@ Watch the first canary or matrix failure, then classify it before rerunning: - **Policy/gate failure:** conflicting labels, invalid changelog, missing authorization, merge conflict, or ineligible artifacts. Correct the gate. GPU reruns will not fix it. - **Superseded run:** a later commit or recognized label change cancelled it through workflow concurrency. Monitor the replacement run rather than reviving stale evidence. -The [`PR Review` workflow](../.github/workflows/claude-pr-review.yml) installs a pinned official Claude Code npm package and checks `claude --version` before passing its executable path to the review action. An installation or startup failure means the review did not run; it is not a review finding or a successful review. Check the installation step before retrying. +The [`Claude Code` workflow](../.github/workflows/claude.yml) has separate review and coding jobs. Review keeps its existing `ready_for_review` and authorized `@pr-claude` triggers, read-only repository contents, and PR feedback permissions; the coding job handles `@claude` and `@Klaud-Cold` with its existing write permissions. Review requests serialize per PR without cancelling active reviews; coding requests remain independent. Both jobs use the pinned official action to install its supported Claude Code CLI. An installation or startup failure means the review did not run; it is not a review finding or a successful review. Check the action installation logs before retrying. ### Rerun safely diff --git a/docs/ci-procedures_zh.md b/docs/ci-procedures_zh.md index 9246ef41a9..a99062214f 100644 --- a/docs/ci-procedures_zh.md +++ b/docs/ci-procedures_zh.md @@ -273,7 +273,7 @@ gh api "/repos/SemiAnalysisAI/InferenceX/actions/runs/$RUN_ID" \ - **策略/Gate 失败:** 标签冲突、Changelog 无效、缺少授权、合并冲突或产物不合格。修正 Gate;重跑 GPU 无法解决。 - **已被取代的 Run:** 后续 Commit 或被识别的标签变更通过 Workflow Concurrency 将其取消。应监控替代 Run,不要复活过期证据。 -[`PR Review` Workflow](../.github/workflows/claude-pr-review.yml) 安装固定版本的官方 Claude Code npm 包,并在将可执行文件路径传给审阅 Action 前检查 `claude --version`。安装或启动失败表示审阅没有执行,既不是代码审阅发现的问题,也不代表审阅通过。重试前应先检查安装步骤。 +[`Claude Code` 工作流](../.github/workflows/claude.yml) 包含审阅和编码两个独立任务。审阅任务保留原有的 `ready_for_review` 及授权 `@pr-claude` 触发条件,只读访问仓库内容,并具有发布 PR 反馈的权限;编码任务使用原有写权限处理 `@claude` 和 `@Klaud-Cold` 请求。同一 PR 的审阅请求串行执行,不取消正在运行的审阅;编码请求仍独立运行。两个任务均通过固定到提交 SHA 的官方 action 安装其支持的 Claude Code CLI。安装或启动失败表示审阅没有执行,既不是代码审阅发现的问题,也不代表审阅通过。重试前应先检查 action 的安装日志。 ### 安全重跑 diff --git a/docs/klaud.md b/docs/klaud.md index e6641e0c83..6040e111af 100644 --- a/docs/klaud.md +++ b/docs/klaud.md @@ -126,7 +126,7 @@ All external actions use full commit SHAs; the table reflects the current workfl ```bash uv run --no-project --exclude-newer PT12H --python 3.12 --with "pydantic>=2.10,<3" python -m infx.klaud --help -uvx zizmor==1.30.0 --offline --no-config --no-ignores .github/workflows/klaud-plan.yml .github/workflows/klaud-candidate.yml +uvx --exclude-newer PT12H zizmor@latest --offline --no-config --no-ignores .github/workflows/klaud-plan.yml .github/workflows/klaud-candidate.yml ``` CLI and workflow checks do not establish GPU workingness. Klaud Cold uses the existing InferenceX validation and e2e workflows for its candidate changes. No live model, benchmark, PR creation or deployment is part of local verification. diff --git a/docs/klaud_zh.md b/docs/klaud_zh.md index 1ee6350e01..da293e1c7b 100644 --- a/docs/klaud_zh.md +++ b/docs/klaud_zh.md @@ -126,7 +126,7 @@ smoke benchmark 和代表性 eval 都通过后,在 changelog 物理末尾追 ```bash uv run --no-project --exclude-newer PT12H --python 3.12 --with "pydantic>=2.10,<3" python -m infx.klaud --help -uvx zizmor==1.30.0 --offline --no-config --no-ignores .github/workflows/klaud-plan.yml .github/workflows/klaud-candidate.yml +uvx --exclude-newer PT12H zizmor@latest --offline --no-config --no-ignores .github/workflows/klaud-plan.yml .github/workflows/klaud-candidate.yml ``` CLI 和工作流检查不能证明 GPU 实际可运行。Klaud Cold 使用现有 InferenceX 校验和 e2e 工作流验证候选修改。本地验证不调用真实模型、不调度 benchmark、不创建 PR、不部署。 diff --git a/docs/testing.md b/docs/testing.md index baf74ed59b..c26398e027 100644 --- a/docs/testing.md +++ b/docs/testing.md @@ -31,7 +31,7 @@ These sources outrank this guide when behavior changes. Update the English page ## Testing layers -[`CI`](../.github/workflows/ci.yml) runs **Lint** and **Tests** in parallel when any `.py` file changes in a PR (including forks) or a push to `main`. GitHub handles path filtering; manual dispatch runs both jobs regardless of changed files. Changes only to docs, shell scripts, YAML, dependencies, or Ruff configuration do not trigger Python CI; run the applicable checks locally or dispatch CI manually. +[`CI`](../.github/workflows/ci.yml) runs **Lint** and **Tests** in parallel for PRs (including forks) and pushes to `main` that change Python files, `ci.yml`, the MCP requirements, Ruff configuration, or `pytest.ini`. [`Workflow security`](../.github/workflows/zizmor.yml) runs **Zizmor** for changes to workflows, action definitions, Dependabot, pre-commit, or zizmor configuration. Python-only changes do not trigger Zizmor; other workflow-only changes do not trigger Lint or Tests. Editing `ci.yml` triggers all three jobs. Each workflow can be dispatched manually. Changes only to other docs, shell scripts, or benchmark YAML do not trigger either workflow; run the applicable checks locally or dispatch them manually. Tests runs every suite under `utils/`, `runners/`, and `experimental/CollectiveX/tests/` with four pytest workers, plus MCP compatibility. New tests in those directories are discovered automatically. The test environment uses Python 3.12 and CPU-only PyTorch; dependencies must be at least 12 hours old. A failing job does not cancel the other; a newer PR update cancels the superseded CI run. Branch pushes without a PR no longer start a separate changelog-test run. @@ -74,6 +74,26 @@ uvx --exclude-newer PT12H ruff@latest format infx Fix findings where practical. Justified exceptions use inline `# noqa: CODE`; unused ignores are checked. Preview rules and automatic unsafe fixes are not enabled. +### GitHub Actions security + +CI uses the latest zizmor release at least 12 hours old, with its strictest `auditor` persona, strict input collection, all supported input kinds, and online action-reference checks. Every unsuppressed finding fails the job, including informational and low-confidence findings. Run the same audit locally with an authenticated GitHub token: + +```bash +GH_TOKEN="$(gh auth token)" uvx --exclude-newer PT12H zizmor@latest \ + --persona auditor --strict-collection --collect all --no-config --no-progress . +``` + +All third-party actions remain pinned to commit SHAs. Same-repository workflow calls use `$/`, which resolves the workflow's exact commit and requires Actions runner 2.336.0 or newer. Dependabot waits seven days before action updates. The combined Claude workflow keeps separate review and coding jobs with their own permissions; the pinned Claude action installs its supported CLI version. + +Auditor mode also reports deliberate architecture choices. Exceptions are attached to the exact affected YAML line with a reason, never disabled globally: + +- Independent GPU dispatches, comment requests, and Klaud waves must not supersede one another. The priority scheduler and candidate ownership claims handle their resource limits. +- Fork sign-off and trusted external dispatch require `pull_request_target`; they execute trusted control code and enforce authorization before privileged operations. +- Existing repository-scoped integration credentials are retained. Moving them into protected GitHub Environments requires migrating the actual stored secrets; adding an empty `environment:` field is not a fix. +- The profiling storage checkout retains its scoped SSH deploy key only because the next step pushes a trace commit to that separate repository. Benchmark checkouts do not retain credentials. + +Add `--no-ignores` to review all of these exceptions. Keep new findings blocking, and review an exception again if its trigger, checkout, credential consumer, or authorization changes. No GPU execution is needed to run this security audit. + ### Parse and syntax ```bash diff --git a/docs/testing_zh.md b/docs/testing_zh.md index aeb497d9a8..1230cb495a 100644 --- a/docs/testing_zh.md +++ b/docs/testing_zh.md @@ -31,7 +31,7 @@ ## 测试层级 -[`CI`](../.github/workflows/ci.yml) 在 PR(包括 fork)或向 `main` 的推送修改任意 `.py` 文件时,并行运行 **Lint** 和 **Tests**。由 GitHub 原生路径筛选决定是否触发;手动分发始终运行两项任务。仅修改文档、Shell 脚本、YAML、依赖或 Ruff 配置不会触发 Python CI;请在本地执行相应检查,或手动分发 CI。 +[`CI`](../.github/workflows/ci.yml) 在 PR(包括 fork)或向 `main` 的推送修改 Python 文件、`ci.yml`、MCP 依赖、Ruff 配置或 `pytest.ini` 时,并行运行 **Lint** 和 **Tests**。[`Workflow security`](../.github/workflows/zizmor.yml) 在工作流、action 定义、Dependabot、pre-commit 或 zizmor 配置变更时运行 **Zizmor**。仅修改 Python 文件不会触发 Zizmor;仅修改其他工作流不会触发 Lint 或 Tests。修改 `ci.yml` 会触发全部三项任务。两个工作流均可手动分发。仅修改其他文档、Shell 脚本或基准测试 YAML 不会触发这两个工作流;请在本地执行相应检查,或手动分发。 Tests 使用四个 pytest worker 运行 `utils/`、`runners/` 和 `experimental/CollectiveX/tests/` 下的全部测试,并检查 MCP 兼容性。这些目录中的新增测试会自动发现。测试环境使用 Python 3.12 和仅支持 CPU 的 PyTorch;依赖必须已发布至少 12 小时。一项任务失败不会取消另一项;PR 更新会取消旧提交的 CI。尚未创建 PR 的分支推送不再单独触发变更日志测试。 @@ -74,6 +74,26 @@ uvx --exclude-newer PT12H ruff@latest format infx 应尽量修复问题。确有理由保留的例外使用行内 `# noqa: CODE`;多余的忽略标记会被检查。未启用预览规则或自动不安全修复。 +### GitHub Actions 安全检查 + +CI 使用发布至少 12 小时的最新 zizmor 版本,启用最严格的 `auditor` 模式、严格输入收集、全部受支持的输入类型,以及在线 action 引用检查。所有未豁免的发现都会使任务失败,包括信息级和低置信度发现。使用已认证的 GitHub token 在本地执行同样的检查: + +```bash +GH_TOKEN="$(gh auth token)" uvx --exclude-newer PT12H zizmor@latest \ + --persona auditor --strict-collection --collect all --no-config --no-progress . +``` + +所有第三方 action 仍固定到提交 SHA。同仓库工作流使用 `$/` 引用,解析到工作流的确切提交,要求 Actions runner 版本至少为 2.336.0。Dependabot 在 action 发布七天后才更新。合并后的 Claude 工作流保留审阅和编码两个独立任务,分别设置权限;固定到提交 SHA 的 Claude action 负责安装其支持的 CLI 版本。 + +Auditor 模式也会报告有意保留的架构选择。豁免仅标注在对应的 YAML 行,并附上原因,不会全局禁用规则: + +- 独立的 GPU 分发、评论请求和 Klaud 批次不应互相取消。资源限制由优先级调度器和候选任务归属声明处理。 +- Fork sign-off 和可信外部分发需要 `pull_request_target`;它们运行可信控制代码,并在执行特权操作前验证授权。 +- 保留现有仓库级集成凭据。迁移到受保护的 GitHub Environments 必须同步迁移实际存储的 secret;仅添加空的 `environment:` 字段不算修复。 +- Profiling 存储仓库的 checkout 保留其专用 SSH deploy key,因为下一步需要向该独立仓库推送 trace 提交。基准测试 checkout 不保留凭据。 + +添加 `--no-ignores` 可复查全部豁免。新增发现仍必须阻止 CI;若豁免涉及的触发器、checkout、凭据使用方或授权发生变化,必须重新审查。该安全检查不需要运行 GPU 任务。 + ### 解析与语法 ```bash