From c7175de1ad5ade7e359ba4ec5eb3f5fca82f8c2e Mon Sep 17 00:00:00 2001 From: Oseltamivir <58582368+Oseltamivir@users.noreply.github.com> Date: Sun, 26 Jul 2026 19:22:43 +0800 Subject: [PATCH] glm5.2-fp4-rtx6000pro-sglang-agentic: add GLM-5.2 SGLang AgentX recipe for RTX PRO 6000 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Adds the SGLang counterpart to the RTX PRO 6000 vLLM recipes: TP8/EP1 GLM-5.2 NVFP4 AgentX replay at concurrency 1/2/4 on the Latitude SM120 node, plus the runner label/launcher routing it needs. SM120 deltas validated on-node: shared-experts fusion must be disabled (the checkpoint keeps the shared expert loose in BF16 while routed experts are packed NVFP4), the sparse DSA attention backend cannot run (its indexer metadata is DeepGEMM-only and DeepGEMM rejects SM120) so the Triton MLA fallback is used, and context is capped at 256k because the FP8 KV pool holds 464,768 tokens per rank. The recipe header records why sparse cannot be re-enabled by flag, so the next reader does not burn a sweep rediscovering it. 中文:新增 GLM-5.2 NVFP4 在 8x RTX PRO 6000 Blackwell(SM120) 节点上的 SGLang 智能体(AgentX)基准配置,TP8/EP1,并发 1/2/4,并补齐该运行器所需的 标签与启动器路由。已在节点上验证的 SM120 差异:必须关闭共享专家融合 (该权重的共享专家为松散 BF16,而路由专家为打包 NVFP4);稀疏 DSA 注意力 后端无法运行(其 indexer 元数据仅有 DeepGEMM 实现,而 DeepGEMM 不支持 SM120),因此回退到 Triton MLA;上下文上限设为 256k,因为 FP8 KV 池每个 rank 仅容纳 464,768 个 token。脚本头部记录了为何不能仅靠开关重新启用稀疏 路径,以避免后续读者重复踩坑并浪费一次扫描。 Co-Authored-By: Claude Opus 5 (1M context) --- .github/workflows/benchmark-tmpl.yml | 7 + .../agentic/glm5.2_fp4_rtx6000pro_sglang.sh | 236 ++++++++++++++++++ configs/nvidia-master.yaml | 32 +++ configs/runners.yaml | 6 + perf-changelog.yaml | 16 ++ runners/launch_rtx6000pro-lat.sh | 144 +++++++++++ 6 files changed, 441 insertions(+) create mode 100755 benchmarks/single_node/agentic/glm5.2_fp4_rtx6000pro_sglang.sh create mode 100755 runners/launch_rtx6000pro-lat.sh diff --git a/.github/workflows/benchmark-tmpl.yml b/.github/workflows/benchmark-tmpl.yml index 0e32953e8a..fe614f7f7e 100644 --- a/.github/workflows/benchmark-tmpl.yml +++ b/.github/workflows/benchmark-tmpl.yml @@ -235,6 +235,13 @@ jobs: done fi + # The direct-host RTX runner mounts this checkout into a rootful + # container. Repair files left by an interrupted job before + # actions/checkout attempts its clean reset. + if [ "${{ runner.name }}" = "rtx6000pro-lat_00" ]; then + sudo -n chown -R "$(id -u):$(id -g)" "$GITHUB_WORKSPACE" + fi + # Cleanup SLURM resources if command -v squeue >/dev/null 2>&1; then echo "[Slurm] Cleaning up jobs with name: ${{ runner.name }} ..." diff --git a/benchmarks/single_node/agentic/glm5.2_fp4_rtx6000pro_sglang.sh b/benchmarks/single_node/agentic/glm5.2_fp4_rtx6000pro_sglang.sh new file mode 100755 index 0000000000..c10643fcfb --- /dev/null +++ b/benchmarks/single_node/agentic/glm5.2_fp4_rtx6000pro_sglang.sh @@ -0,0 +1,236 @@ +#!/usr/bin/env bash +set -euo pipefail +set -x + +# AgentX trace replay for GLM-5.2 NVFP4 on the 8x RTX PRO 6000 Blackwell +# (SM120) node using SGLang. +# +# Flags start from the merged B300 NVFP4 cookbook recipe +# (benchmarks/single_node/agentic/glm5.2_fp4_b300_sglang.sh, STP only) and +# then apply the SM120 deltas this GPU family needs: +# +# * TP8 only. The checkpoint is 433 GB, so pure TP across all eight 96 GB +# GPUs is the only layout that leaves room for the KV pool; TP4 does not +# fit the weights at all. +# * Shared-experts fusion is force-disabled. SGLang enables it for this +# config (n_routed_experts=256, n_shared_experts=1, EP off), but +# nvidia/GLM-5.2-NVFP4 stores the shared expert loose and unquantized +# (BF16 [2048, 6144]) while the routed experts are packed NVFP4 +# ([2048, 3072]), so the fused loader aborts during weight load with +# "The size of tensor a (3072) must match the size of tensor b (6144)". +# This is the same trap the in-tree comment records for the +# compressed-tensors Kimi-K2.5 checkpoint. +# * Attention falls back to Triton MLA. SGLang would default this DSA model +# to --attention-backend dsa, whose indexer metadata comes only from +# DeepGEMM, and DeepGEMM has no SM120 kernel. Sparse attention is +# therefore not exercised on this GPU family, and prefill cost grows +# superlinearly with context. +# * MoE runner backend is left at SGLang's own SM120 choice for +# modelopt_fp4 (flashinfer_cutlass; trtllm-gen MoE is SM100-only) and is +# overridable through MOE_RUNNER_BACKEND. +# * HiCache is not supported on RTX PRO 6000, so this recipe is +# GPU-resident KV only. +# +# Required env vars: +# MODEL, TP, CONC, KV_OFFLOADING, TOTAL_CPU_DRAM_GB, RESULT_DIR, DURATION, +# EP_SIZE, DP_ATTENTION + +source "$(dirname "$0")/../../benchmark_lib.sh" + +check_env_vars \ + MODEL \ + TP \ + CONC \ + KV_OFFLOADING \ + TOTAL_CPU_DRAM_GB \ + RESULT_DIR \ + DURATION \ + EP_SIZE \ + DP_ATTENTION + +if [[ "$TP" != "8" ]]; then + echo "GLM-5.2 SGLang on RTX PRO 6000 requires TP8: the 433 GB NVFP4 checkpoint does not fit in fewer than eight 96 GB GPUs" >&2 + exit 1 +fi +if [[ "$KV_OFFLOADING" != "none" ]]; then + echo "GLM-5.2 SGLang on RTX PRO 6000 supports GPU-resident KV cache only (HiCache is unsupported on this GPU family)" >&2 + exit 1 +fi +if [[ "$DP_ATTENTION" == "true" ]]; then + echo "GLM-5.2 SGLang on RTX PRO 6000 does not support DP attention: attention-DP replicates the KV pool per rank, which the 96 GB GPUs cannot hold alongside 54 GB of weights" >&2 + exit 1 +fi +if [[ "$EP_SIZE" != "1" ]]; then + echo "GLM-5.2 SGLang on RTX PRO 6000 supports EP1 only" >&2 + exit 1 +fi + +# `hf download` creates the target dir if missing and is itself idempotent. +# When MODEL_PATH is unset (stand-alone runs), fall back to the HF_HUB_CACHE. +# Either way, MODEL_PATH is what the server is launched with. +MODEL_REVISION="${GLM52_MODEL_REVISION:-aec724e8c7b8ee9db3b48c01c320f63f9cdaf8aa}" +if [[ -n "${MODEL_PATH:-}" ]]; then + if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then + hf download "$MODEL" --revision "$MODEL_REVISION" --local-dir "$MODEL_PATH" + fi +else + hf download "$MODEL" --revision "$MODEL_REVISION" + export MODEL_PATH="$MODEL" +fi + +if [[ -n "${SLURM_JOB_ID:-}" ]]; then + echo "JOB $SLURM_JOB_ID running on ${SLURMD_NODENAME:-unknown}" +fi +nvidia-smi +nvidia-smi topo -m || true + +# The FP8 KV pool holds 464,768 tokens per rank (~51 KB/token across 78 +# layers) at mem-fraction-static 0.85, so the model's 1M context does not fit +# and the server is capped at 256k. Replay the matching 256k-capped corpus +# instead of the 1M default this model prefix would otherwise select. +export WEKA_LOADER_OVERRIDE="${WEKA_LOADER_OVERRIDE:-semianalysis_cc_traces_weka_062126_256k}" +resolve_trace_source +install_agentic_deps + +SERVER_LOG="$RESULT_DIR/server.log" +mkdir -p "$RESULT_DIR" + +export PYTHONNOUSERSITE=1 +export TORCH_CUDA_ARCH_LIST=12.0a +# All eight GPUs are SYS-connected (PCIe only, no NVLink) on this node, and +# NCCL 2.28.9 segfaults probing its bnxt_re devices; the runner already sets +# NCCL_IB_DISABLE=1. Keep collectives on the local PCIe/SHM transports. +export NCCL_P2P_LEVEL="${NCCL_P2P_LEVEL:-SYS}" +export NCCL_PROTO="${NCCL_PROTO:-LL,LL128,Simple}" +export GLOO_SOCKET_IFNAME="${GLOO_SOCKET_IFNAME:-lo}" +export NCCL_SOCKET_IFNAME="${NCCL_SOCKET_IFNAME:-lo}" +export OMP_NUM_THREADS="${OMP_NUM_THREADS:-16}" + +# NOTE for whoever revisits the DSA path: SGLang carries a set of SM120 +# kernel fixups (SGLANG_OPT_FP8_WO_A_GEMM / SGLANG_OPT_USE_TOPK_V2 / +# SGLANG_OPT_USE_TILELANG_MHC_PRE / SGLANG_OPT_DEEPGEMM_HC_PRENORM off, +# SGLANG_FP8_PAGED_MQA_LOGITS_TORCH on) that it applies only inside its +# DeepseekV4ForCausalLM branch, and GLM-5.2 (GlmMoeDsaForCausalLM) misses +# them. They were tested on-node and are NOT needed on this Triton MLA +# fallback path — boot and generation are identical with and without them — +# so they are deliberately not exported here. They become relevant again if +# the sparse DSA backend ever works on SM120. +# +# Do not simply switch --attention-backend back to dsa: enabling the sparse +# path on SM120 was attempted on-node (2026-07-26) and needs kernel work, not +# a flag. Patching SGLang to route the indexer's paged-MQA logits through its +# TileLang/torch implementations (both already used by the dsv4 indexer) and +# to stop building the DeepGEMM schedule plan does clear the "Unsupported +# architecture" abort, and the server then boots and serves — but four more +# walls follow: +# 1. TileLang's CUDA sparse-MLA kernel takes bf16 KV only (its fp8 variants +# are ROCm-only), which halves the KV pool to 256,256 tokens/rank. +# 2. That kernel asks for 170,048 B of dynamic shared memory; SM120 allows +# ~100 KB. Its tile size is not a knob — the barrier arrive_counts +# (384/256/128) encode the BI=64 thread mapping, so block_I=32 compiles +# and then reads out of bounds. +# 3. The plain (v1) kernel does fit at block_I=32 / num_stages=1 / 128 +# threads, but CUDA-graph capture fails because TileLang JIT-compiles +# inside the capture (cudaErrorStreamCaptureUnsupported). +# 4. With graphs disabled it runs and is still numerically wrong: degenerate +# repetition on a 26-token prompt and an empty answer on the 45k-token +# needle, identically with the TileLang and the torch reference logits +# kernels — so the fault is the sparse-MLA kernel itself, not the indexer. +# Fixing this means an SM120-shaped sparse-MLA kernel (split the d_v=512 +# accumulation so KV tiles stay under ~32 KB, re-derive the barrier counts, +# validate against the dense path), not a config change. Full logs from the +# attempt: rtx6000pro-lat:/home/ubuntu/glm52-sglang-patch/out/server.dsa*.log. + +# Agentic warmup dispatches hundreds of large prompts at once; allow up to +# 15 minutes of TCP progress before AIPerf declares a connection dead. +export AIPERF_HTTP_TCP_USER_TIMEOUT=900000 +# AIPerf pins one pooled keep-alive connection per session (client-side +# keep-alive 300s) while uvicorn's default SGLANG_TIMEOUT_KEEP_ALIVE is 5s; +# inter-turn idle gaps can reuse a socket exactly as the server closes it -> +# ECONNRESET -> terminal warmup failure. Outlast the client pool. +export SGLANG_TIMEOUT_KEEP_ALIVE=900 +# Dense-attention prefill on this node measured 872 tok/s at 4.4k context +# falling to 418 tok/s at 244k (single stream), so a warmup snapshot of +# 100k-token histories needs several minutes per trajectory. Double the +# shared 1800s warmup grace so warmup drains instead of being declared +# failed; grace is a maximum wait, not a fixed sleep. +export AGENTIC_WARMUP_GRACE_PERIOD="${AGENTIC_WARMUP_GRACE_PERIOD:-3600}" + +# AgentX concurrency counts live session trees, not individual requests. +# Allow subagent fan-out to exceed CONC without clipping request bursts. +MAX_RUNNING_REQUESTS=$((2 * CONC)) +CUDA_GRAPH_MAX_BS=$MAX_RUNNING_REQUESTS +[ "$CUDA_GRAPH_MAX_BS" -gt 64 ] && CUDA_GRAPH_MAX_BS=64 + +MEM_FRACTION_STATIC="${MEM_FRACTION_STATIC:-0.85}" +CONTEXT_LENGTH="${CONTEXT_LENGTH:-262144}" +ATTENTION_BACKEND="${ATTENTION_BACKEND:-triton}" + +MOE_ARGS=() +if [[ -n "${MOE_RUNNER_BACKEND:-}" ]]; then + MOE_ARGS=(--moe-runner-backend "$MOE_RUNNER_BACKEND") +fi + +SGLANG_CMD=( + python3 -m sglang.launch_server + --model-path "$MODEL_PATH" + --served-model-name "$MODEL" + --host 0.0.0.0 + --port "$PORT" + --trust-remote-code + --tp "$TP" + --ep-size "$EP_SIZE" + --quantization modelopt_fp4 + # nvidia/GLM-5.2-NVFP4 keeps the shared expert loose and in BF16, so the + # fused-shared-expert loader cannot consume it (see the header note). + --disable-shared-experts-fusion + # SGLang would default this DSA model to --attention-backend dsa, whose + # indexer metadata is built by deep_gemm.get_paged_mqa_logits_metadata + # (the only CUDA option; 'cutedsl' is gated to SM100). DeepGEMM aborts + # with "Assertion error (attention.hpp:227): Unsupported architecture" on + # SM120 during warmup, so fall back to Triton MLA. Sparse attention is + # consequently not exercised on this GPU family. + --attention-backend "$ATTENTION_BACKEND" + "${MOE_ARGS[@]}" + # GLM-5.2 emits the GLM-4.7-style // + # format; glm47 is required for structured message.tool_calls, and the + # reasoning parser keeps hybrid-thinking output in reasoning_content. + --tool-call-parser glm47 + --reasoning-parser glm45 + --context-length "$CONTEXT_LENGTH" + --chunked-prefill-size 8192 + --mem-fraction-static "$MEM_FRACTION_STATIC" + --max-running-requests "$MAX_RUNNING_REQUESTS" + --cuda-graph-max-bs "$CUDA_GRAPH_MAX_BS" + --watchdog-timeout 1800 + --enable-metrics +) + +printf '%q ' "${SGLANG_CMD[@]}" | tee "$RESULT_DIR/sglang_command.txt" +printf '\n' | tee -a "$RESULT_DIR/sglang_command.txt" + +echo "Starting SGLang server for RTX PRO 6000..." +"${SGLANG_CMD[@]}" > "$SERVER_LOG" 2>&1 & +SERVER_PID=$! +echo "Server PID: $SERVER_PID" + +cleanup_agentic_server() { + local exit_code=$? + trap - EXIT INT TERM + set +e + stop_background_process_tree "$SERVER_PID" "SGLang server" 60 + exit "$exit_code" +} +trap cleanup_agentic_server EXIT +trap 'exit 130' INT +trap 'exit 143' TERM + +wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" + +if [ "${EVAL_ONLY:-false}" = "true" ]; then + run_eval --port "$PORT" +else + build_replay_cmd "$RESULT_DIR" + REPLAY_CMD+=" --server-metrics http://localhost:$PORT/metrics" + run_agentic_replay_and_write_outputs "$RESULT_DIR" +fi diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index a388cad95d..e99835f57a 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -8097,3 +8097,35 @@ glm5.2-fp4-b300-sglang-agentic: # radix cache (GPU hit 0.93->0.57 at 48->64) and thrashes on re-prefill, so it is # strictly dominated by conc 48 on both throughput and interactivity. - { tp: 8, ep: 8, dp-attn: true, kv-offloading: dram, kv-offload-backend: { name: hicache }, conc-list: [48], router: { name: sglang-router, version: "0.3.2" } } + +# GLM-5.2 NVFP4 AgentX on the 8x RTX PRO 6000 Blackwell (SM120) Latitude node +# with SGLang. Three SM120 facts shape this entry: +# +# * TP8 only. The checkpoint is 433 GB (56 GB/GPU loaded), so pure TP across +# all eight 96 GB GPUs is the only layout that leaves room for a KV pool; +# TP4 cannot hold the weights, and attention-DP would replicate the pool +# per rank. +# * Sparse DSA attention is unavailable. SGLang's DSA backend builds its +# indexer paged-MQA-logits metadata through DeepGEMM (the only CUDA +# option; 'cutedsl' is gated to SM100), and DeepGEMM asserts "Unsupported +# architecture" on SM120, so the recipe runs the Triton MLA fallback with +# the sparse indexer bypassed. Prefill therefore scales superlinearly with +# context (measured on-node: 17.6k tokens 5.2 s -> 70.5k tokens 38.9 s). +# * Context is capped at 256k, not the model's 1M. The FP8 KV pool is +# 464,768 tokens per rank at mem-fraction-static 0.85 (~51 KB/token across +# 78 layers), so the recipe replays the 256k-capped trace corpus and stops +# at conc 4, where four full-length sessions already exceed the pool and +# start re-prefilling. +glm5.2-fp4-rtx6000pro-sglang-agentic: + image: lmsysorg/sglang:v0.5.15.post1-cu130 + model: nvidia/GLM-5.2-NVFP4 + model-prefix: glm5.2 + runner: cluster:rtx6000pro-lat + precision: fp4 + framework: sglang + multinode: false + scenarios: + agentic-coding: + - dram-utilization: 0.80 + search-space: + - { tp: 8, ep: 1, dp-attn: false, kv-offloading: none, conc-list: [1, 2, 4] } diff --git a/configs/runners.yaml b/configs/runners.yaml index 851b821ba2..5008248a1b 100644 --- a/configs/runners.yaml +++ b/configs/runners.yaml @@ -164,6 +164,10 @@ labels: - gb300-nv_0 - gb300-nv_1 - gb300-nv_2 + rtx6000pro: + - rtx6000pro-lat_00 + rtx6000pro-lat: + - rtx6000pro-lat_00 cluster:h100-cw: - h100-cw_00 - h100-cw_01 @@ -251,6 +255,8 @@ labels: - gb300-nv_0 - gb300-nv_1 - gb300-nv_2 + cluster:rtx6000pro-lat: + - rtx6000pro-lat_00 cluster:mi300x-amds: - mi300x-amds_00 - mi300x-amds_01 diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 975c917743..19c05b8007 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -5066,3 +5066,19 @@ description: - "Bump SGLang container image from lmsysorg/sglang:v0.5.12-cu130 to lmsysorg/sglang:v0.5.15.post1-cu130 (https://github.com/sgl-project/sglang/releases/tag/v0.5.15.post1)" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2313 + +- config-keys: + - glm5.2-fp4-rtx6000pro-sglang-agentic + scenario-type: + - agentic-coding + description: + - "Add a GLM-5.2 NVFP4 SGLang AgentX sweep on the 8x RTX PRO 6000 Blackwell (SM120) Latitude runner, TP8/EP1 with GPU-resident FP8 KV cache at concurrency 1, 2, and 4" + - "Pin lmsysorg/sglang:v0.5.15.post1-cu130, the same image as the merged B300 GLM-5.2 SGLang recipe and the first release whose GlmMoeDsaForCausalLM support is present" + - "Force --disable-shared-experts-fusion: SGLang enables fusion for this config, but nvidia/GLM-5.2-NVFP4 stores the shared expert loose in BF16 while the routed experts are packed NVFP4, so the fused loader aborts weight load with a 3072-vs-6144 shape mismatch" + - "Run the Triton MLA attention fallback instead of the sparse DSA backend: SGLang builds DSA indexer paged-MQA-logits metadata through DeepGEMM only ('cutedsl' is gated to SM100), and DeepGEMM asserts 'Unsupported architecture' on SM120, so GLM-5.2's sparse-attention advantage is not realized on this GPU family" + - "Do not carry SGLang's DeepseekV4-only SM120 kernel fixups (FP8 weight-only GEMM, topk_v2, tilelang/DeepGEMM hidden-compression prenorm, torch paged-MQA logits): tested on-node, boot and generation are identical with and without them on this Triton fallback path" + - "Cap context at 256k and replay the 256k-capped trace corpus: the FP8 KV pool holds 464,768 tokens per rank at mem-fraction-static 0.85 (~51 KB/token across 78 layers), so the model's 1M context does not fit and conc >4 re-prefills continuously" + - "Raise the agentic warmup grace period to 3600s: dense-attention prefill measured 872 tok/s at 4.4k context falling to 418 tok/s at 244k, so long-history warmup snapshots need more than the shared 1800s default" + - "Validated over SSH: TP8 boot on the node (56.04 GB weights/GPU, 464,768-token KV pool/rank, CUDA graphs captured, /health 200), short-prompt correctness through the glm45/glm47 parsers, and exact needle retrieval at 45,394 / 155,552 / 220,029-token prompts with no NaN from the SM120 flashinfer_cutlass NVFP4 MoE path" + - "Record in the recipe why the sparse DSA path cannot be re-enabled by flag: patching SGLang to use its TileLang/torch paged-MQA-logits kernels clears the DeepGEMM abort and boots, but the TileLang sparse-MLA kernel needs bf16 KV, asks for 170,048 B of shared memory against SM120's ~100 KB, cannot be CUDA-graph captured, and is numerically wrong at the only tile size that fits" + pr-link: XXX diff --git a/runners/launch_rtx6000pro-lat.sh b/runners/launch_rtx6000pro-lat.sh new file mode 100755 index 0000000000..4b7b0e7548 --- /dev/null +++ b/runners/launch_rtx6000pro-lat.sh @@ -0,0 +1,144 @@ +#!/usr/bin/bash +set -euo pipefail + +# This runner executes directly on the single RTX PRO 6000 GPU node. Docker +# therefore owns a separate image cache from the node's RKE2/containerd cache. +HF_HUB_CACHE_MOUNT="${HF_HUB_CACHE_MOUNT:-/var/lib/inferencex/hf-hub-cache}" +export HF_HUB_CACHE="${HF_HUB_CACHE:-/mnt/hf_hub_cache/}" +PORT="${PORT:-8888}" + +# NCCL 2.28.9 segfaults while probing this node's bnxt_re RDMA devices. +# Disable that RDMA path by default while preserving local CUDA P2P/SHM and +# allowing an explicit caller override. +export NCCL_IB_DISABLE="${NCCL_IB_DISABLE:-1}" + +: "${GITHUB_WORKSPACE:?GITHUB_WORKSPACE must be set}" +: "${IMAGE:?IMAGE must be set}" +: "${EXP_NAME:?EXP_NAME must be set}" +: "${PRECISION:?PRECISION must be set}" + +mkdir -p "$HF_HUB_CACHE_MOUNT" + +export GPU_COUNT="${GPU_COUNT:-${TP:?TP must be set}}" +if [[ ! "$GPU_COUNT" =~ ^[1-9][0-9]*$ ]]; then + echo "GPU_COUNT must be a positive integer, got: $GPU_COUNT" >&2 + exit 1 +fi + +export CUDA_VISIBLE_DEVICES +CUDA_VISIBLE_DEVICES="$(seq -s, 0 "$((GPU_COUNT - 1))")" + +# Some Slurm/enroot configs spell registry paths as nvcr.io#namespace/image. +# Docker requires the normal slash form. +DOCKER_IMAGE="${IMAGE//#//}" + +SPEC_SUFFIX="" +if [[ "${SPEC_DECODING:-}" == "mtp" ]]; then + SPEC_SUFFIX="_mtp" +fi + +export SCENARIO_SUBDIR="${SCENARIO_SUBDIR:-fixed_seq_len/}" +SCENARIO_SUBDIR="${SCENARIO_SUBDIR#/}" +SCENARIO_SUBDIR="${SCENARIO_SUBDIR%/}/" +BENCH_BASE="benchmarks/single_node/${SCENARIO_SUBDIR}${EXP_NAME%%_*}_${PRECISION}_rtx6000pro" + +# Prefer the framework-tagged name so several engines can serve the same model +# on this runner, and fall back to the untagged name that predates it. +BENCH_SCRIPT="${BENCH_BASE}_${FRAMEWORK:-vllm}${SPEC_SUFFIX}.sh" +if [[ ! -f "$GITHUB_WORKSPACE/$BENCH_SCRIPT" ]]; then + BENCH_SCRIPT="${BENCH_BASE}${SPEC_SUFFIX}.sh" +fi + +if [[ ! -f "$GITHUB_WORKSPACE/$BENCH_SCRIPT" ]]; then + echo "Benchmark script not found: $GITHUB_WORKSPACE/$BENCH_SCRIPT" >&2 + exit 1 +fi + +server_name="bmk-server-${RUNNER_NAME:-rtx6000pro-lat}" +server_name="${server_name//[^a-zA-Z0-9_.-]/-}" + +cleanup() { + docker rm -f "$server_name" >/dev/null 2>&1 || true +} +trap cleanup EXIT + +# Clear a container left behind by a cancelled or interrupted workflow. +cleanup + +docker run \ + --rm \ + --pull=missing \ + --name="$server_name" \ + --runtime=nvidia \ + --gpus="$GPU_COUNT" \ + --network=host \ + --ipc=host \ + --privileged \ + --shm-size=32g \ + --ulimit memlock=-1 \ + --ulimit stack=67108864 \ + --security-opt seccomp=unconfined \ + --cap-add=SYS_PTRACE \ + --volume "$HF_HUB_CACHE_MOUNT:$HF_HUB_CACHE" \ + --volume "$GITHUB_WORKSPACE:/workspace/" \ + --workdir=/workspace/ \ + --env HF_TOKEN \ + --env HF_HUB_CACHE \ + --env MODEL \ + --env MODEL_PREFIX \ + --env MODEL_PATH \ + --env TP \ + --env PP_SIZE \ + --env DCP_SIZE \ + --env PCP_SIZE \ + --env EP_SIZE \ + --env DP_SIZE \ + --env DP_ATTENTION \ + --env GPU_COUNT \ + --env CONC \ + --env MAX_MODEL_LEN \ + --env ISL \ + --env OSL \ + --env FRAMEWORK \ + --env PRECISION \ + --env DISAGG \ + --env SPEC_DECODING \ + --env NUM_SPEC_TOKENS \ + --env RUN_EVAL \ + --env EVAL_ONLY \ + --env EVAL_LIMIT \ + --env EVAL_MAX_MODEL_LEN \ + --env RUNNER_TYPE \ + --env RUNNER_NAME \ + --env RESULT_FILENAME \ + --env RESULT_DIR \ + --env RANDOM_RANGE_RATIO \ + --env GPU_MEM_UTIL \ + --env AIPERF_FAILED_REQUEST_THRESHOLD \ + --env KV_OFFLOADING \ + --env KV_OFFLOAD_BACKEND \ + --env KV_OFFLOAD_BACKEND_METADATA \ + --env ROUTER_METADATA \ + --env KV_P2P_TRANSFER \ + --env TOTAL_CPU_DRAM_GB \ + --env DURATION \ + --env SCENARIO_TYPE \ + --env SCENARIO_SUBDIR \ + --env IS_AGENTIC \ + --env SWEBENCH_GEN_MODE \ + --env SWEBENCH_USE_MODAL \ + --env MODAL_TOKEN_ID \ + --env MODAL_TOKEN_SECRET \ + --env PROFILE \ + --env SGLANG_TORCH_PROFILER_DIR \ + --env VLLM_TORCH_PROFILER_DIR \ + --env VLLM_RPC_TIMEOUT \ + --env PYTHONDONTWRITEBYTECODE \ + --env PYTHONPYCACHEPREFIX=/tmp/pycache/ \ + --env PORT="$PORT" \ + --env CUDA_DEVICE_ORDER=PCI_BUS_ID \ + --env CUDA_VISIBLE_DEVICES \ + --env NCCL_IB_DISABLE \ + --entrypoint=/bin/bash \ + "$DOCKER_IMAGE" \ + "$BENCH_SCRIPT"