diff --git a/inferencex-e2e/benchmarks/multi_node/agentic/dsv4_fp4_mi355x_atom-disagg.sh b/inferencex-e2e/benchmarks/multi_node/agentic/dsv4_fp4_mi355x_atom-disagg.sh deleted file mode 100755 index 931434e1a7..0000000000 --- a/inferencex-e2e/benchmarks/multi_node/agentic/dsv4_fp4_mi355x_atom-disagg.sh +++ /dev/null @@ -1,130 +0,0 @@ -#!/usr/bin/env bash - -# Agentic trace-replay recipe for a disaggregated ATOM server on MI355X -# (DeepSeek-V4-Pro FP4, 1P1D TP8), mooncake RDMA KV transfer + atomesh router. -# -# CI-style sibling of the former SGLang dsv4_fp4_mi355x_sglang-disagg.sh (same -# agentic trace workload, same submit.sh path), but drives the ATOM engine. -# Modeled on ATOM recipes/DeepSeek-V4-Agentic-PD-Max.md: three concurrency -# tiers selected by the search space -- TP (conc 1-32), DP-attention -# (conc 64-128, no offload), and DP-attention + CPU KV offload (conc 256, -# lmcache_offload multi connector). Per-tier server behavior lives in -# amd_utils/server_atom.sh (IS_AGENTIC branch) and models_atom.yaml -# (DeepSeek-V4-Pro-AgentX). - -SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" -source "$SCRIPT_DIR/../../benchmark_lib.sh" - -check_env_vars \ - CONC_LIST \ - ISL \ - OSL \ - IMAGE \ - SPEC_DECODING \ - MODEL_PATH \ - PREFILL_NUM_WORKERS \ - PREFILL_TP \ - PREFILL_EP \ - PREFILL_DP_ATTN \ - DECODE_NUM_WORKERS \ - DECODE_TP \ - DECODE_EP \ - DECODE_DP_ATTN \ - PREFILL_NODES \ - DECODE_NODES \ - RANDOM_RANGE_RATIO \ - DURATION \ - KV_OFFLOADING \ - IS_AGENTIC \ - FRAMEWORK - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -set -x - -# Use upstreamed multi_node scripts (no external clone needed) -cd "$GITHUB_WORKSPACE/benchmarks/multi_node/amd_utils" || exit 1 - -# Set up ATOM launch script-specific environment variables -export TIME_LIMIT="${TIME_LIMIT:-08:00:00}" -export MODEL_PATH=$MODEL_PATH -export MODEL_NAME=$MODEL_NAME -export CONTAINER_IMAGE=$IMAGE - -# ── Identity / result naming ── -export MODEL_PREFIX="${MODEL_PREFIX:-dsv4}" -export PRECISION="${PRECISION:-fp4}" -export RESULT_FILENAME="${RESULT_FILENAME:-${RUNNER_NAME:-dsv4-fp4-agentic}}" - -# ── Agentic benchmark params ── -export DURATION="${DURATION:-1800}" -# DSV4-Pro max model len for agentic traces (matches single-node recipe). -export MAX_MODEL_LEN="${MAX_MODEL_LEN:-1000000}" - -# ── KV cache offloading (ATOM lmcache_offload CPU tier) ── -# KV_OFFLOADING=none | dram (passed from YAML; none for the TP/DP tiers, dram -# for the conc-256 offload tier). KV_OFFLOAD_BACKEND selects the backend when -# offloading is on; the ATOM PD path only implements the lmcache_offload CPU -# tier, so "lmcache" is the only supported value. The multi-connector JSON and -# per-rank sizing are built in server_atom.sh from TOTAL_CPU_DRAM_GB (aggregate -# budget from the matrix, dram-utilization 0.80). -export KV_OFFLOADING="${KV_OFFLOADING:-none}" -if [[ "$KV_OFFLOADING" != "none" ]]; then - export KV_OFFLOAD_BACKEND="${KV_OFFLOAD_BACKEND:-lmcache}" - # Recipe (DeepSeek-V4-Agentic-PD-Max.md, "The offload settings that matter"): - # both default to values this workload cannot live with. Prefill node only. - export OFFLOAD_SLOT_STAGING_SLOTS="${OFFLOAD_SLOT_STAGING_SLOTS:-4}" - export OFFLOAD_COPY_WORKERS="${OFFLOAD_COPY_WORKERS:-1}" - export OFFLOAD_MIN_LOAD_TOKENS="${OFFLOAD_MIN_LOAD_TOKENS:-8192}" -fi - -# ── MTP ── -# EAGLE/MTP synthetic acceptance length on agentic throughput runs (real target -# verification is used only on eval-only runs). 2.49 per the PD-Max recipe. -export DECODE_MTP_SIZE="${DECODE_MTP_SIZE:-0}" -export SPEC_DECODE_AL="${SPEC_DECODE_AL:-2.49}" - -# Derive EP/DP enable flags from the topology inputs. -if [[ "${PREFILL_EP:-1}" -eq 1 ]]; then -export PREFILL_ENABLE_EP=false -else -export PREFILL_ENABLE_EP=true -fi - -if [[ "$PREFILL_DP_ATTN" == "true" ]]; then -export PREFILL_ENABLE_DP=true -else -export PREFILL_ENABLE_DP=false -fi - -if [[ "${DECODE_EP:-1}" -eq 1 ]]; then -export DECODE_ENABLE_EP=false -else -export DECODE_ENABLE_EP=true -fi - -if [[ "$DECODE_DP_ATTN" == "true" ]]; then -export DECODE_ENABLE_DP=true -else -export DECODE_ENABLE_DP=false -fi - -# Launch the job. CONC_LIST is space-delimited in YAML; submit.sh wants 'x'. -JOB_ID=$(bash ./submit.sh $PREFILL_NODES \ - $PREFILL_NUM_WORKERS \ - $DECODE_NODES \ - $DECODE_NUM_WORKERS \ - $ISL $OSL "${CONC_LIST// /x}" inf \ - ${PREFILL_ENABLE_EP} ${PREFILL_ENABLE_DP} \ - ${DECODE_ENABLE_EP} ${DECODE_ENABLE_DP} \ - ${PREFILL_TP} ${DECODE_TP} \ - ${RANDOM_RANGE_RATIO}) - -if [[ $? -ne 0 ]]; then - echo "Failed to submit job" >&2 - exit 1 -fi - -echo "$JOB_ID" diff --git a/inferencex-e2e/benchmarks/multi_node/amd_utils/env_atom.sh b/inferencex-e2e/benchmarks/multi_node/amd_utils/env_atom.sh deleted file mode 100755 index 71cbdf06ff..0000000000 --- a/inferencex-e2e/benchmarks/multi_node/amd_utils/env_atom.sh +++ /dev/null @@ -1,40 +0,0 @@ -#!/bin/bash -# ATOM/mooncake environment, sourced by server_atom.sh in place of env.sh. -# IBDEVICES: RDMA device names (e.g. ionic_0,ionic_1,...), set by the runner or -# auto-detected. - -set -x - -export PYTHONUNBUFFERED=1 -export PYTHONDONTWRITEBYTECODE=1 - - -if [[ -z "$IBDEVICES" ]]; then - DETECTED=$(ibv_devinfo 2>/dev/null | grep "hca_id:" | awk '{print $2}' | paste -sd',') - if [[ -n "$DETECTED" ]]; then - export IBDEVICES="$DETECTED" - echo "[INFO] Auto-detected IBDEVICES=$IBDEVICES via ibv_devinfo on $(hostname -s)" - else - # ATOM passes no IB device to the server (mooncake picks its own RDMA device via - # proxy_ip/handshake_port), so a missing IBDEVICES is non-fatal here. - echo "[WARN] Unable to detect RDMA devices via ibv_devinfo; IBDEVICES unset (non-fatal for ATOM/mooncake)" >&2 - fi -else - echo "[INFO] Using IBDEVICES=$IBDEVICES (set by runner or environment)" -fi -export IBDEVICES - - -export LD_LIBRARY_PATH=/opt/venv/lib/python3.10/site-packages/mooncake:/opt/rocm/lib:${LD_LIBRARY_PATH:-} - -export SAFETENSORS_FAST_GPU=1 - -export VLLM_LOG_LEVEL=WARNING -export ATOM_LOG_LEVEL=WARNING -export AITER_LOG_LEVEL=WARNING -export LOG_LEVEL=WARNING -export LOGLEVEL=WARNING - -set +x - -echo "[INFO] ATOM env: IBDEVICES=$IBDEVICES LD_LIBRARY_PATH includes mooncake" \ No newline at end of file diff --git a/inferencex-e2e/benchmarks/multi_node/amd_utils/job.slurm b/inferencex-e2e/benchmarks/multi_node/amd_utils/job.slurm index 455e9d3cb0..881ca6a017 100755 --- a/inferencex-e2e/benchmarks/multi_node/amd_utils/job.slurm +++ b/inferencex-e2e/benchmarks/multi_node/amd_utils/job.slurm @@ -34,9 +34,7 @@ echo "" # Use $(pwd) not BASH_SOURCE — sbatch copies the script to /var/spool/slurmd/ # at runtime, but the CWD remains the submit-time directory (amd_utils/). -if [[ "$ENGINE" == "atom-disagg" ]]; then - MODELS_YAML="$(pwd)/models_atom.yaml" -elif [[ "$ENGINE" == "tilert" ]]; then +if [[ "$ENGINE" == "tilert" ]]; then MODELS_YAML="$(pwd)/models_tilert.yaml" else MODELS_YAML="$(pwd)/models.yaml" @@ -543,26 +541,7 @@ DOCKER_ENV_COMMON=( ) # Engine-specific env vars -if [[ "$ENGINE" == "atom-disagg" ]]; then - check_env_vars \ - PREFILL_PORT DECODE_PORT HANDSHAKE_PORT MEM_FRAC_STATIC KV_CACHE_DTYPE \ - BLOCK_SIZE MAX_NUM_SEQS - DOCKER_ENV_ENGINE=( - -e ATOM_WS_PATH=${WS_PATH} - -e PREFILL_PORT=${PREFILL_PORT} - -e DECODE_PORT=${DECODE_PORT} - -e ROUTER_PORT=${ROUTER_PORT} - -e HANDSHAKE_PORT=${HANDSHAKE_PORT} - -e MEM_FRAC_STATIC=${MEM_FRAC_STATIC} - -e KV_CACHE_DTYPE=${KV_CACHE_DTYPE} - -e BLOCK_SIZE=${BLOCK_SIZE} - -e MAX_NUM_SEQS=${MAX_NUM_SEQS} - -e MAX_MODEL_LEN=${MAX_MODEL_LEN:-} - -e MAX_NUM_BATCHED_TOKENS=${MAX_NUM_BATCHED_TOKENS:-} - -e EXTRA_SERVER_ARGS=\${EXTRA_SERVER_ARGS:-} - -e IBDEVICES=${IBDEVICES:-} - ) -elif [[ "$ENGINE" == "tilert" ]]; then +if [[ "$ENGINE" == "tilert" ]]; then DOCKER_ENV_ENGINE=( -e MODEL_PATH=$DOCKER_MODEL_PATH -e PREFILL_IMAGE=${PREFILL_IMAGE} @@ -671,84 +650,6 @@ echo \"Rank \$SLURM_PROCID on \$(hostname)\" eval \"\$DOCKER_CMD_DETECT\" echo \"[docker-detect] rank \$SLURM_PROCID: DOCKER_CMD=\$DOCKER_CMD\" -# Enable out-of-tree RDMA library mounts for atom-disagg (mooncake requires host RDMA stack) -RDMA_MOUNTS=() -if [[ "$ENGINE" == "atom-disagg" ]]; then - -# When the container base OS differs from the host (e.g. Ubuntu 24.04 image -# on a 22.04 host), the container's bundled libibverbs/libionic may be -# ABI-incompatible with the host kernel drivers. Detect the NIC type and -# bind-mount the host's out-of-tree RDMA userspace libraries into the -# container so the RDMA stack always matches the running kernel. -_detect_nic_type() { - if [[ -n \"\${MORI_NIC_TYPE:-}\" ]]; then echo \"\$MORI_NIC_TYPE\"; return; fi - local bnxt=0 mlx5=0 ionic=0 - if [[ -d /sys/class/infiniband ]]; then - for dev in /sys/class/infiniband/*; do - local name; name=\$(basename \"\$dev\") - case \"\$name\" in - bnxt_re*) ((bnxt++)) ;; mlx5*) ((mlx5++)) ;; ionic*) ((ionic++)) ;; - *) - local drv; drv=\$(basename \"\$(readlink -f \"\$dev/device/driver\" 2>/dev/null)\" 2>/dev/null || true) - case \"\$drv\" in bnxt*) ((bnxt++)) ;; mlx5*) ((mlx5++)) ;; ionic*) ((ionic++)) ;; esac ;; - esac - done - fi - if (( bnxt >= mlx5 && bnxt >= ionic && bnxt > 0 )); then echo bnxt - elif (( ionic >= mlx5 && ionic > 0 )); then echo ionic - else echo mlx5; fi -} - -_find_host_ibverbs() { - for c in /usr/lib64/libibverbs.so.1 /lib/x86_64-linux-gnu/libibverbs.so.1 /usr/lib/x86_64-linux-gnu/libibverbs.so.1.14.39.0 /usr/lib/x86_64-linux-gnu/libibverbs.so.1; do - local r; r=\$(readlink -f \"\$c\" 2>/dev/null || true) - [[ \"\$r\" == *libibverbs.so.1.14.57.0 ]] && continue - if [[ -f \"\$r\" ]]; then echo \"\$r\"; return; fi - done -} - -_NIC_TYPE=\$(_detect_nic_type) -echo \"[rdma] NIC type: \${_NIC_TYPE} on \$(hostname)\" - -if [[ \"\$_NIC_TYPE\" == \"ionic\" || \"\$_NIC_TYPE\" == \"bnxt\" ]]; then - _host_ibv=\$(_find_host_ibverbs) - if [[ -n \"\$_host_ibv\" ]]; then - RDMA_MOUNTS+=(-v \"\$_host_ibv:/lib/x86_64-linux-gnu/libibverbs.so.1\") - fi -fi - -if [[ \"\$_NIC_TYPE\" == \"ionic\" ]]; then - for _dir in /usr/local/lib /usr/lib/x86_64-linux-gnu; do - for _lib in \"\$_dir\"/libionic*.so; do - [[ -f \"\$_lib\" ]] || continue - _real=\$(readlink -f \"\$_lib\") - [[ -f \"\$_real\" ]] && RDMA_MOUNTS+=(-v \"\$_real:\$_real\") - RDMA_MOUNTS+=(-v \"\$_lib:/usr/lib/x86_64-linux-gnu/\$(basename \"\$_lib\")\") - done - done - if [[ -d /usr/lib/x86_64-linux-gnu/libibverbs ]]; then - for _lib in /usr/lib/x86_64-linux-gnu/libibverbs/libionic-rdmav*.so; do - [[ -f \"\$_lib\" ]] && RDMA_MOUNTS+=(-v \"\$_lib:\$_lib\") - done - fi - [[ -d /etc/libibverbs.d ]] && RDMA_MOUNTS+=(-v /etc/libibverbs.d:/etc/libibverbs.d:ro) -elif [[ \"\$_NIC_TYPE\" == \"bnxt\" ]]; then - for _lib in /usr/local/lib/libbnxt_re-rdmav*.so; do - [[ -f \"\$_lib\" ]] && RDMA_MOUNTS+=(-v \"\$_lib:/usr/lib/x86_64-linux-gnu/libibverbs/\$(basename \"\$_lib\")\") - done - for _lib in /usr/local/lib/libbnxt_re.so; do - [[ -f \"\$_lib\" ]] && RDMA_MOUNTS+=(-v \"\$_lib:/usr/lib/x86_64-linux-gnu/\$(basename \"\$_lib\")\") - done - [[ -d /etc/libibverbs.d ]] && RDMA_MOUNTS+=(-v /etc/libibverbs.d:/etc/libibverbs.d:ro) -fi - -if [[ \${#RDMA_MOUNTS[@]} -gt 0 ]]; then - echo \"[rdma] bind-mounts: \${RDMA_MOUNTS[*]}\" -else - echo \"[rdma] no out-of-tree RDMA mounts needed\" -fi -fi # end: if ENGINE == atom-disagg - RANK_IMAGE= if [[ \"$ENGINE\" == \"tilert\" && \"\$SLURM_PROCID\" -lt \"$xP\" ]]; then RANK_IMAGE=\"$PREFILL_IMAGE\" @@ -790,7 +691,6 @@ exec \$DOCKER_CMD run \ -v ${HICACHE_MC_CONFIG}:/config/hicache_mc.env:ro \ ${EXTRA_DOCKER_MOUNTS:-} \ ${CLIENT_DOCKER_MOUNTS} \ - \${RDMA_MOUNTS[@]+"\${RDMA_MOUNTS[@]}"} \ ${DOCKER_ENV_COMMON[*]} \ ${DOCKER_ENV_ENGINE[*]} \ ${CLIENT_DOCKER_ENV} \ diff --git a/inferencex-e2e/benchmarks/multi_node/amd_utils/models_atom.yaml b/inferencex-e2e/benchmarks/multi_node/amd_utils/models_atom.yaml deleted file mode 100644 index 008edc2116..0000000000 --- a/inferencex-e2e/benchmarks/multi_node/amd_utils/models_atom.yaml +++ /dev/null @@ -1,83 +0,0 @@ -# Model-specific ATOM server configurations for disaggregated inference. -# -# Each top-level key is a MODEL_NAME value (must match the directory name under MODEL_DIR). -# -# To add a new model: add a new top-level entry following the same schema. -# No script changes are required. -# -# Schema: -# : -# env: str # Space-separated KEY=VALUE pairs exported unconditionally -# tp_dp_flags: str # Shared TP+DPA flags (fallback when prefill/decode-specific keys are absent) -# prefill_tp_dp_flags: str # TP+DPA flags for prefill only (overrides tp_dp_flags) -# decode_tp_dp_flags: str # TP+DPA flags for decode only (overrides tp_dp_flags) -# tp_dp_env: str # Space-separated KEY=VALUE pairs exported only in TP+DPA mode -# ep_dp_flags: str # Shared EP+DPA flags (fallback when prefill/decode-specific keys are absent) -# prefill_ep_dp_flags: str # EP+DPA flags for prefill only (overrides ep_dp_flags) -# decode_ep_dp_flags: str # EP+DPA flags for decode only (overrides ep_dp_flags) -# ep_dp_env: str # Space-separated KEY=VALUE pairs exported only in EP+DPA mode -# mtp_flags: str # Flags passed to SPEC_ARGS before $DECODE_MTP_SIZE (e.g. "--method mtp --num-speculative-tokens") -# kv_cache_flags: str # Full --kv_cache_dtype flag string (e.g. "--kv_cache_dtype fp8", or "" for none) -# online_quant_config: str # JSON string passed to --online_quant_config (used when DPA is disabled) -# online_quant_dpa_config: str # JSON string passed to --online_quant_config when DPA is enabled (falls back to online_quant_config) -# block_size: str # --block-size value (overrides server_atom.sh default of 16) -# mem_frac_static: str # --gpu-memory-utilization value (overrides default of 0.85) -# max_model_len: str # --max-model-len value (overrides default of unset) -# max_num_seqs: str # --max-num-seqs value (overrides default of 256) -# max_num_batched_tokens: str # --max-num-batched-tokens value (overrides default of unset) -# scheduler_delay_factor: str # --scheduler-delay-factor value (overrides default of unset) -# Agentic-only (applied by server_atom.sh only when IS_AGENTIC=1): -# attn_prefill_chunk_size: str # --attn-prefill-chunk-size value (unset = omit) -# state_checkpoint_interval_tokens: str # --state-checkpoint-interval-tokens value (unset = omit) -# level: str # --level value (unset = omit) -# spec_decode_acceptance_length: str # --spec-decode-acceptance-length (synthetic AL on throughput runs; unset = omit) - -# Agentic (AgentX trace-replay) variant of DeepSeek-V4-Pro on ATOM PD. Resolved -# by job.slurm/server_atom.sh as '-AgentX' when IS_AGENTIC=1. Differs -# from the throughput entry above per ATOM recipes/DeepSeek-V4-Agentic-PD-Max.md: -# prefix caching (server_atom.sh IS_AGENTIC branch), FP8 KV/index cache, TBO on -# prefill only, agentic prefill-chunk / state-checkpoint / level knobs, and MTP -# with synthetic acceptance on throughput runs. -DeepSeek-V4-Pro-AgentX: &DeepSeek-V4-Pro-AgentX - # Rail-isolated RDMA: matched rails replace dest-device affinity. Omit - # ATOM_MOONCAKE_MATCHED_RAILS if every default P/D HCA pair is mutually - # reachable on the fabric. - # ATOM_NUMA_BIND and the ATOM_DP_* ports are unconditional here: the recipe - # exports them in every tier, including TP (concurrency 1-32) where DP - # attention is off, so they cannot live in tp_dp_env/ep_dp_env (DP-only). - env: "ATOM_MOE_GU_ITLV=1 AITER_BF16_FP8_MOE_BOUND=0 MC_GID_INDEX=1 ATOM_MOONCAKE_MATCHED_RAILS=auto NCCL_IB_DISABLE=1 ATOM_DISABLE_MMAP=true ATOM_NUMA_BIND=1 ATOM_DP_MASTER_PORT=29510 ATOM_DP_BASE_PORT=29610 ATOM_PREFIX_CACHE_POLICY=lru ATOM_PREFIX_CACHE_PROTECTED_RATIO=0.5" - kv_cache_flags: "--kv_cache_dtype fp8 --index-cache-dtype fp4" - # DP-attention tiers (conc 64+): TBO is prefill-only per the recipe. - tp_dp_flags: "--enable-dp-attention" - prefill_tp_dp_flags: "--enable-dp-attention --enable-tbo" - decode_tp_dp_flags: "--enable-dp-attention" - ep_dp_flags: "--enable-expert-parallel --enable-dp-attention" - prefill_ep_dp_flags: "--enable-expert-parallel --enable-dp-attention --enable-tbo" - decode_ep_dp_flags: "--enable-expert-parallel --enable-dp-attention" - # DP-only env for the agentic DP-attention tiers (prefill and decode). Under - # --dp-aware the cache-aware router sends explicit P/D rank hints that take - # priority over engine-local session affinity and load balancing, so - # ATOM_DP_SESSION_AFFINITY and ATOM_DP_LB_REQ_EQUIV are not needed on this - # routing path (recipe). Cross-DP prefill coalescing (ATOM_ENABLE_PREFILL_DELAYER=0) - # is off on both roles per the recipe. NUMA binding and the DP ports moved to - # the unconditional env above because the recipe sets them in the TP tier too. - tp_dp_env: "ATOM_ENABLE_PREFILL_DELAYER=0" - ep_dp_env: "ATOM_ENABLE_PREFILL_DELAYER=0" - # Prefill-only DP env: the recipe sets GPU_MAX_HW_QUEUES=5 on the DP-attention - # prefill node and leaves decode unset. server_atom.sh applies this on prefill - # nodes only. - prefill_dp_env: "GPU_MAX_HW_QUEUES=5" - mtp_flags: "--method dspark --num-speculative-tokens" - # config.py forces block-size 256 for V4 (lcm(4,128)); 16 is ignored/misleading. - block_size: "256" - mem_frac_static: "0.9" - max_num_batched_tokens: "16384" - # Agentic-only server knobs consumed by server_atom.sh IS_AGENTIC branch. - attn_prefill_chunk_size: "16384" - state_checkpoint_interval_tokens: "8192" - level: "3" - spec_decode_acceptance_length: "3.01" - -# The PD-Max recipe runs DSpark, which needs the draft head bundled in the -# -0813 config.json. Same alias split as models.yaml on the SGLang side. -DeepSeek-V4-Pro-0813-AgentX: *DeepSeek-V4-Pro-AgentX diff --git a/inferencex-e2e/benchmarks/multi_node/amd_utils/server.sh b/inferencex-e2e/benchmarks/multi_node/amd_utils/server.sh index f304f53b40..a3830b57aa 100755 --- a/inferencex-e2e/benchmarks/multi_node/amd_utils/server.sh +++ b/inferencex-e2e/benchmarks/multi_node/amd_utils/server.sh @@ -4,7 +4,6 @@ source "$(dirname "${BASH_SOURCE[0]}")/../../benchmark_lib.sh" --validation-only # Multi-Engine Disaggregated Server Dispatcher # Dispatches to the engine-specific server launcher based on ENGINE env var. # ENGINE=sglang-disagg (default) -> server_sglang.sh (SGLang + MoRI) -# ENGINE=atom-disagg -> server_atom.sh (ATOM + mooncake) # ENGINE=tilert -> server_tilert.sh (vLLM prefill + TileRT decode) check_env_vars ENGINE WS_PATH @@ -17,10 +16,7 @@ export WS_PATH ENGINE echo "[DISPATCHER] ENGINE=$ENGINE WS_PATH=$WS_PATH" -if [[ "$ENGINE" == "atom-disagg" ]]; then - export ATOM_WS_PATH="$WS_PATH" - source "$WS_PATH/server_atom.sh" -elif [[ "$ENGINE" == "tilert" ]]; then +if [[ "$ENGINE" == "tilert" ]]; then source "$WS_PATH/server_tilert.sh" else source "$WS_PATH/server_sglang.sh" diff --git a/inferencex-e2e/benchmarks/multi_node/amd_utils/server_atom.sh b/inferencex-e2e/benchmarks/multi_node/amd_utils/server_atom.sh deleted file mode 100755 index b7351bebef..0000000000 --- a/inferencex-e2e/benchmarks/multi_node/amd_utils/server_atom.sh +++ /dev/null @@ -1,702 +0,0 @@ -#!/bin/bash -# ATOM disaggregated launcher: mooncake RDMA KV transfer and atomesh routing. - -source "$(dirname "${BASH_SOURCE[0]}")/../../benchmark_lib.sh" --validation-only -check_env_vars \ - MODEL_NAME ROUTER_PORT PREFILL_PORT DECODE_PORT HANDSHAKE_PORT \ - MEM_FRAC_STATIC BLOCK_SIZE MAX_NUM_SEQS WAIT_SERVER_TIMEOUT - -check_env_vars \ - NODE0_ADDR NODE_RANK xP yD IPADDRS \ - PREFILL_TP_SIZE DECODE_TP_SIZE PREFILL_ENABLE_EP PREFILL_ENABLE_DP DECODE_ENABLE_EP \ - DECODE_ENABLE_DP DECODE_MTP_SIZE BENCH_INPUT_LEN BENCH_OUTPUT_LEN BENCH_RANDOM_RANGE_RATIO \ - BENCH_REQUEST_RATE BENCH_NUM_PROMPTS_MULTIPLIER BENCH_MAX_CONCURRENCY DRY_RUN GPUS_PER_NODE \ - RUN_EVAL EVAL_ONLY EVAL_FRAMEWORK BENCHMARK_LOGS_DIR MODEL_DIR \ - ATOM_WS_PATH - -EXTRA_SERVER_ARGS="${EXTRA_SERVER_ARGS:-}" - -source $ATOM_WS_PATH/setup_deps.sh -source $ATOM_WS_PATH/env_atom.sh - -# lm-eval with high num_concurrent exhausts the default 1024 FD limit. -ulimit -n 65536 2>/dev/null || ulimit -n 8192 2>/dev/null || true -echo "ulimit -n (open files): $(ulimit -n)" - -host_ip=$(ip route get 1.1.1.1 2>/dev/null | awk '/src/ {print $7}') -if [[ -z "$host_ip" ]]; then - host_ip=$(hostname -I 2>/dev/null | awk '{print $1}') -fi -host_name=$(hostname) - -# ATOM/mooncake handshake IP: the recipe exports this per node (prefill/decode -# IP). Default to this node's resolved IP so it matches the mooncake proxy_ip. -export ATOM_HOST_IP="${ATOM_HOST_IP:-$host_ip}" - -set -x -_yaml_tmp=$(mktemp) -python3 << PYEOF > "$_yaml_tmp" -import yaml -# Resolve the recipe entry the same way server_sglang.sh does: agentic runs -# (IS_AGENTIC) use the '-AgentX' entry, non-agentic runs use the bare -# ''. job.slurm passes MODEL_NAME unchanged (base name), so the -AgentX -# derivation has to happen here. Fall back to the base entry when absent. -with open('${ATOM_WS_PATH}/models_atom.yaml') as f: - _all = yaml.safe_load(f) or {} -_name = '${MODEL_NAME}' -_agentic = '${IS_AGENTIC:-0}'.strip().lower() in ('1', 'true') -_key = f'{_name}-AgentX' if _agentic else _name -m = _all.get(_key, _all.get(_name, {})) -import sys -print(f"Selected models_atom.yaml entry: {_key if _key in _all else _name} (IS_AGENTIC={_agentic})", file=sys.stderr) -def sh(v): return v.replace("'", "'\\''") -print(f"MODEL_ENVS='{sh(m.get('env', ''))}'") -_tp_dp = m.get('tp_dp_flags', '') -print(f"PREFILL_MODEL_TP_DP_FLAGS='{sh(m.get('prefill_tp_dp_flags', _tp_dp))}'") -print(f"DECODE_MODEL_TP_DP_FLAGS='{sh(m.get('decode_tp_dp_flags', _tp_dp))}'") -_ep_dp = m.get('ep_dp_flags', '') -print(f"PREFILL_MODEL_EP_DP_FLAGS='{sh(m.get('prefill_ep_dp_flags', _ep_dp))}'") -print(f"DECODE_MODEL_EP_DP_FLAGS='{sh(m.get('decode_ep_dp_flags', _ep_dp))}'") -print(f"MODEL_TP_DP_ENV='{sh(m.get('tp_dp_env', ''))}'") -print(f"MODEL_EP_DP_ENV='{sh(m.get('ep_dp_env', ''))}'") -print(f"MODEL_PREFILL_DP_ENV='{sh(m.get('prefill_dp_env', ''))}'") -print(f"MODEL_MTP_FLAGS='{sh(m.get('mtp_flags', ''))}'") -print(f"MODEL_KV_ARG='{sh(m.get('kv_cache_flags', ''))}'") -print(f"_ONLINE_QUANT_CONFIG='{sh(m.get('online_quant_config', ''))}'") -print(f"_ONLINE_QUANT_DPA_CONFIG='{sh(m.get('online_quant_dpa_config', m.get('online_quant_config', '')))}'") -print(f"_YAML_BLOCK_SIZE='{sh(m.get('block_size', ''))}'") -print(f"_YAML_MEM_FRAC_STATIC='{sh(m.get('mem_frac_static', ''))}'") -print(f"_YAML_MAX_MODEL_LEN='{sh(m.get('max_model_len', ''))}'") -print(f"_YAML_MAX_NUM_SEQS='{sh(m.get('max_num_seqs', ''))}'") -print(f"_YAML_MAX_NUM_BATCHED_TOKENS='{sh(m.get('max_num_batched_tokens', ''))}'") -print(f"_YAML_SCHEDULER_DELAY_FACTOR='{sh(m.get('scheduler_delay_factor', ''))}'") -print(f"_YAML_ATTN_PREFILL_CHUNK_SIZE='{sh(m.get('attn_prefill_chunk_size', ''))}'") -print(f"_YAML_STATE_CKPT_INTERVAL='{sh(m.get('state_checkpoint_interval_tokens', ''))}'") -print(f"_YAML_LEVEL='{sh(m.get('level', ''))}'") -print(f"_YAML_SPEC_DECODE_AL='{sh(m.get('spec_decode_acceptance_length', ''))}'") -PYEOF -# shellcheck source=/dev/null -source "$_yaml_tmp" -rm -f "$_yaml_tmp" -unset _yaml_tmp - -# Model YAML overrides the caller-provided server tuning. -BLOCK_SIZE="${_YAML_BLOCK_SIZE:-${BLOCK_SIZE}}" -MEM_FRAC_STATIC="${_YAML_MEM_FRAC_STATIC:-${MEM_FRAC_STATIC}}" -MAX_MODEL_LEN="${_YAML_MAX_MODEL_LEN:-${MAX_MODEL_LEN:-}}" -MAX_NUM_SEQS="${_YAML_MAX_NUM_SEQS:-${MAX_NUM_SEQS}}" -MAX_NUM_BATCHED_TOKENS="${_YAML_MAX_NUM_BATCHED_TOKENS:-${MAX_NUM_BATCHED_TOKENS:-}}" -SCHEDULER_DELAY_FACTOR="${_YAML_SCHEDULER_DELAY_FACTOR:-${SCHEDULER_DELAY_FACTOR:-}}" -ATTN_PREFILL_CHUNK_SIZE="${_YAML_ATTN_PREFILL_CHUNK_SIZE:-}" -STATE_CKPT_INTERVAL="${_YAML_STATE_CKPT_INTERVAL:-}" -LEVEL="${_YAML_LEVEL:-}" -# Synthetic acceptance length: YAML > launcher env (SPEC_DECODE_AL). -SPEC_DECODE_AL="${_YAML_SPEC_DECODE_AL:-${SPEC_DECODE_AL:-}}" -unset _YAML_BLOCK_SIZE _YAML_MEM_FRAC_STATIC _YAML_MAX_MODEL_LEN _YAML_MAX_NUM_SEQS _YAML_MAX_NUM_BATCHED_TOKENS _YAML_SCHEDULER_DELAY_FACTOR -unset _YAML_ATTN_PREFILL_CHUNK_SIZE _YAML_STATE_CKPT_INTERVAL _YAML_LEVEL _YAML_SPEC_DECODE_AL - -# ============================================================================= -# Agentic (AgentX trace-replay) run configuration -# ============================================================================= -# Agentic runs (IS_AGENTIC) use the '-AgentX' recipe and differ from the -# throughput path: prefix caching on, per-request max-num-seqs = 2*conc, extra -# ATOM server knobs, dp-sticky router, and an optional CPU KV-offload tier on -# prefill. All of this is gated on IS_AGENTIC_RUN so the throughput path is -# unchanged. Reference: ATOM recipes/DeepSeek-V4-Agentic-PD-Max.md. -IS_AGENTIC_RUN=0 -if [[ "${IS_AGENTIC:-0}" == "1" || "${IS_AGENTIC:-}" == "true" ]]; then - IS_AGENTIC_RUN=1 -fi - -# Largest concurrency in this allocation (BENCH_MAX_CONCURRENCY is x-delimited). -_MAX_CONC=$(echo "$BENCH_MAX_CONCURRENCY" | tr 'x' '\n' | sort -n | tail -1) - -# Prefix caching: agentic runs depend on cross-turn prefix reuse; throughput -# runs keep the server's paged-only behavior. -if [[ "$IS_AGENTIC_RUN" == "1" ]]; then - PREFIX_CACHE_ARG="--enable-prefix-caching" - # Recipe max-num-seqs is 2*concurrency (prefill and decode). - MAX_NUM_SEQS=$((2 * _MAX_CONC)) -else - PREFIX_CACHE_ARG="--no-enable_prefix_caching" -fi - -# Agentic-only server knobs (applied when the model provides them). -AGENTIC_SERVER_ARGS="" -if [[ "$IS_AGENTIC_RUN" == "1" ]]; then - [[ -n "$ATTN_PREFILL_CHUNK_SIZE" ]] && AGENTIC_SERVER_ARGS+=" --attn-prefill-chunk-size ${ATTN_PREFILL_CHUNK_SIZE}" - [[ -n "$STATE_CKPT_INTERVAL" ]] && AGENTIC_SERVER_ARGS+=" --state-checkpoint-interval-tokens ${STATE_CKPT_INTERVAL}" - [[ -n "$LEVEL" ]] && AGENTIC_SERVER_ARGS+=" --level ${LEVEL}" - AGENTIC_SERVER_ARGS+=" --cudagraph-mode FULL" -fi - -IFS=',' read -ra IP_ARRAY <<< "$IPADDRS" - -PREFILL_NODES_PER_WORKER=$(((PREFILL_TP_SIZE + GPUS_PER_NODE - 1) / GPUS_PER_NODE)) -DECODE_NODES_PER_WORKER=$(((DECODE_TP_SIZE + GPUS_PER_NODE - 1) / GPUS_PER_NODE)) -NODE_OFFSET=$((PREFILL_NODES_PER_WORKER * xP)) - -PREFILL_ARGS="" -PREFILL_IPS=() -for i in $(seq 0 $((xP - 1))); do - idx=$((i * PREFILL_NODES_PER_WORKER)) - PREFILL_IPS[$i]="${IP_ARRAY[$idx]}" - PREFILL_ARGS="$PREFILL_ARGS --prefill http://${IP_ARRAY[$idx]}:${PREFILL_PORT}" -done - -DECODE_ARGS="" -DECODE_IPS=() -for i in $(seq 0 $((yD - 1))); do - idx=$((i * DECODE_NODES_PER_WORKER + NODE_OFFSET)) - DECODE_IPS[$i]="${IP_ARRAY[$idx]}" - DECODE_ARGS="$DECODE_ARGS --decode http://${IP_ARRAY[$idx]}:${DECODE_PORT}" -done - -PREFILL_PARALLEL_ARGS=(-tp "$PREFILL_TP_SIZE") #TP -ONLINE_QUANT_ARG="" -if [ "$PREFILL_ENABLE_DP" = "true" ]; then - if [ "$PREFILL_ENABLE_EP" = "true" ]; then #EP+DPA - PREFILL_PARALLEL_ARGS=(-tp "$PREFILL_TP_SIZE" ${PREFILL_MODEL_EP_DP_FLAGS}) - for _dp_env_pair in ${MODEL_EP_DP_ENV}; do export "$_dp_env_pair"; done - else #TP+DPA - PREFILL_PARALLEL_ARGS=(-tp "$PREFILL_TP_SIZE" ${PREFILL_MODEL_TP_DP_FLAGS}) - for _dp_env_pair in ${MODEL_TP_DP_ENV}; do export "$_dp_env_pair"; done - fi - if [[ -n "$_ONLINE_QUANT_DPA_CONFIG" ]]; then - ONLINE_QUANT_ARG="--online_quant_config '${_ONLINE_QUANT_DPA_CONFIG}'" - fi -else - if [[ -n "$_ONLINE_QUANT_CONFIG" ]]; then - ONLINE_QUANT_ARG="--online_quant_config '${_ONLINE_QUANT_CONFIG}'" - fi -fi - -DECODE_PARALLEL_ARGS=(-tp "$DECODE_TP_SIZE") #TP -if [ "$DECODE_ENABLE_DP" = "true" ]; then - if [ "$DECODE_ENABLE_EP" = "true" ]; then #EP+DPA - DECODE_PARALLEL_ARGS=(-tp "$DECODE_TP_SIZE" ${DECODE_MODEL_EP_DP_FLAGS}) - for _dp_env_pair in ${MODEL_EP_DP_ENV}; do export "$_dp_env_pair"; done - else #TP+DPA - DECODE_PARALLEL_ARGS=(-tp "$DECODE_TP_SIZE" ${DECODE_MODEL_TP_DP_FLAGS}) - for _dp_env_pair in ${MODEL_TP_DP_ENV}; do export "$_dp_env_pair"; done - fi -fi -# Prefill-only DP env (e.g. GPU_MAX_HW_QUEUES): the shared DP env above is -# exported on every node, so scope prefill-only knobs by role here. NODE_RANK < -# NODE_OFFSET is a prefill node (see the node-role branch below); the recipe -# leaves these unset on decode. -if [ "$PREFILL_ENABLE_DP" = "true" ] && [ "$NODE_RANK" -lt "$NODE_OFFSET" ]; then - for _dp_env_pair in ${MODEL_PREFILL_DP_ENV}; do export "$_dp_env_pair"; done -fi -unset _dp_env_pair -unset _ONLINE_QUANT_CONFIG _ONLINE_QUANT_DPA_CONFIG - -for _env_pair in ${MODEL_ENVS}; do - export "$_env_pair" -done -unset _env_pair - -SPEC_ARGS=() -if [[ -n "$MODEL_MTP_FLAGS" && "${DECODE_MTP_SIZE}" -gt 0 ]]; then - SPEC_ARGS=(${MODEL_MTP_FLAGS} "$DECODE_MTP_SIZE") - # Agentic throughput runs simulate acceptance at the recipe's synthetic AL; - # eval runs (RUN_EVAL / EVAL_ONLY) need real target verification, so skip it. - if [[ "$IS_AGENTIC_RUN" == "1" && -n "$SPEC_DECODE_AL" \ - && "${EVAL_ONLY:-false}" != "true" && "${RUN_EVAL:-false}" != "true" ]]; then - SPEC_ARGS+=(--spec-decode-acceptance-length "$SPEC_DECODE_AL") - fi -fi - -KV_CACHE_ARG="${MODEL_KV_ARG}" - -MODEL_LEN_ARGS="" -if [[ -n "$MAX_MODEL_LEN" ]]; then - MODEL_LEN_ARGS="${MODEL_LEN_ARGS} --max-model-len ${MAX_MODEL_LEN}" -fi -if [[ -n "$MAX_NUM_BATCHED_TOKENS" ]]; then - MODEL_LEN_ARGS="${MODEL_LEN_ARGS} --max-num-batched-tokens ${MAX_NUM_BATCHED_TOKENS}" -fi -if [[ -n "$SCHEDULER_DELAY_FACTOR" ]]; then - MODEL_LEN_ARGS="${MODEL_LEN_ARGS} --scheduler-delay-factor ${SCHEDULER_DELAY_FACTOR}" -fi - -# ============================================================================= -# PD KV-transfer connectors and router policy -# ============================================================================= -# Decode is always a plain mooncake consumer. Prefill is a plain mooncake -# producer, except on the agentic CPU-offload tier (KV_OFFLOADING=dram) where it -# wraps mooncake + lmcache_offload in a "multi" connector (recipe -# DeepSeek-V4-Agentic-PD-Max.md, "DP attention with CPU offload"). host_ip is -# this node's handshake IP (resolved above). -DECODE_KV_TRANSFER="{\"kv_role\":\"kv_consumer\",\"kv_connector\":\"mooncake\",\"proxy_ip\":\"${host_ip}\",\"handshake_port\":${HANDSHAKE_PORT}}" -PREFILL_KV_TRANSFER="{\"kv_role\":\"kv_producer\",\"kv_connector\":\"mooncake\",\"proxy_ip\":\"${host_ip}\",\"handshake_port\":${HANDSHAKE_PORT}}" -if [[ "$IS_AGENTIC_RUN" == "1" && "${KV_OFFLOADING:-none}" == "dram" ]]; then - # lmcache.max_local_cpu_size is per worker; TOTAL_CPU_DRAM_GB is the - # aggregate CPU budget from the matrix (dram-utilization), so divide by - # GPUS_PER_NODE (one offload worker per GPU rank). - _per_worker_cpu_gb=$(( ${TOTAL_CPU_DRAM_GB:-0} / GPUS_PER_NODE )) - if [[ "$_per_worker_cpu_gb" -le 0 ]]; then _per_worker_cpu_gb=128; fi - # Recipe offload env (prefill node only). These are read by the lmcache - # offload runtime, not encoded in the connector JSON, so they must be in the - # server process env -- the launcher exports them outside the SLURM/Docker - # boundary where they are lost, so set them here. PYTHONHASHSEED=0 keeps the - # LMCache prefix hashes consistent across the offload worker processes. - export PYTHONHASHSEED="${PYTHONHASHSEED:-0}" - export OFFLOAD_COPY_WORKERS="${OFFLOAD_COPY_WORKERS:-1}" - export OFFLOAD_MIN_LOAD_TOKENS="${OFFLOAD_MIN_LOAD_TOKENS:-8192}" - export OFFLOAD_SLOT_STAGING_SLOTS="${OFFLOAD_SLOT_STAGING_SLOTS:-4}" - PREFILL_KV_TRANSFER="{\"kv_connector\":\"multi\",\"connectors\":[{\"kv_role\":\"kv_producer\",\"kv_connector\":\"mooncake\",\"proxy_ip\":\"${host_ip}\",\"handshake_port\":${HANDSHAKE_PORT}},{\"kv_connector\":\"lmcache_offload\",\"kv_role\":\"offload\",\"offload_layout\":\"hybrid\",\"max_pending_saves\":8,\"slot_sidecar_staging_slots\":${OFFLOAD_SLOT_STAGING_SLOTS:-4},\"lmcache.local_cpu\":true,\"lmcache.max_local_cpu_size\":${_per_worker_cpu_gb},\"lmcache.local_disk\":null,\"lmcache.max_local_disk_size\":0,\"lmcache.remote_url\":null,\"lmcache.chunk_size\":256,\"lmcache.cache_policy\":\"LRU\",\"lmcache.lookup_server_worker_ids\":[],\"lmcache.store_location\":\"LocalCPUBackend\",\"lmcache.retrieve_locations\":[\"LocalCPUBackend\"]}]}" -fi - -# Router policy: the agentic DP-attention tiers route cache-aware with balance -# thresholds, the agentic TP tier routes round-robin, and both pin PD rank -# mapping to none. Throughput runs keep random. -ROUTER_POLICY_ARGS="--policy random" -if [[ "$IS_AGENTIC_RUN" == "1" ]]; then - if [[ "$PREFILL_ENABLE_DP" == "true" ]]; then - if [[ "$_MAX_CONC" -eq 256 ]]; then - # conc=256: pin each session to a fixed DP rank (dp_sticky) and let - # aiperf derive the session id from the request correlation id so the - # router can keep the sticky mapping. - ROUTER_POLICY_ARGS="--dp-aware --prefill-policy dp_sticky --decode-policy dp_sticky --atom-pd-rank-mapping-policy none" - export AIPERF_HTTP_X_SESSION_ID_FROM_CORRELATION_ID=1 - else - ROUTER_POLICY_ARGS="--dp-aware --prefill-policy cache_aware --decode-policy cache_aware --cache-threshold 0.8 --balance-abs-threshold 20 --balance-rel-threshold 2.0 --eviction-interval 300 --atom-pd-rank-mapping-policy none" - fi - else - ROUTER_POLICY_ARGS="--prefill-policy round_robin --decode-policy round_robin --atom-pd-rank-mapping-policy none" - fi -fi - -cat < prefill node 0 + router; 1..NODE_OFFSET-1 -> prefill; -# NODE_OFFSET.. -> decode. -if [ "$NODE_RANK" -eq 0 ]; then - echo "NODE INFO =======================================" - echo "${host_name}:${host_ip} is Prefill Node 0 + Router" - echo "Prefill TP=${PREFILL_TP_SIZE}, Decode TP=${DECODE_TP_SIZE}" - echo "Prefill servers: ${PREFILL_ARGS}" - echo "Decode servers: ${DECODE_ARGS}" - echo "================================================" - - PREFILL_CMD="python3 -m atom.entrypoints.openai_server \ - --model ${MODEL_DIR}/${MODEL_NAME} \ - --host 0.0.0.0 --server-port ${PREFILL_PORT} \ - --trust-remote-code \ - ${PREFILL_PARALLEL_ARGS[*]} \ - ${SPEC_ARGS[*]} \ - ${KV_CACHE_ARG} \ - --block-size ${BLOCK_SIZE} \ - --gpu-memory-utilization ${MEM_FRAC_STATIC} \ - --max-num-seqs ${MAX_NUM_SEQS} \ - ${MODEL_LEN_ARGS} \ - ${AGENTIC_SERVER_ARGS} \ - ${PREFIX_CACHE_ARG} \ - ${ONLINE_QUANT_ARG} \ - --kv-transfer-config '${PREFILL_KV_TRANSFER}' \ - ${EXTRA_SERVER_ARGS}" - - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: $PREFILL_CMD" - else - set -x - eval "$PREFILL_CMD" \ - 2>&1 | tee /run_logs/slurm_job-${SLURM_JOB_ID}/prefill0_${host_name}.log & - set +x - prefill0_pid=$! - fi - - echo "[-------]" NODE $NODE_RANK "[--------]" - echo "Waiting for all servers to be up (timeout=${WAIT_SERVER_TIMEOUT}s)..." - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: wait for prefill/decode /health endpoints" - else - _deadline=$(( $(date +%s) + WAIT_SERVER_TIMEOUT )) - for _ip in "${PREFILL_IPS[@]}"; do - echo "[wait] prefill http://${_ip}:${PREFILL_PORT}/health" - while ! curl -sf --max-time 10 "http://${_ip}:${PREFILL_PORT}/health" >/dev/null 2>&1; do - if [[ $(date +%s) -ge $_deadline ]]; then - echo "[wait][FAIL] prefill ${_ip}:${PREFILL_PORT} not ready after ${WAIT_SERVER_TIMEOUT}s" >&2 - exit 1 - fi - sleep 10 - done - echo "[wait][OK] prefill ${_ip}:${PREFILL_PORT} ready" - done - for _ip in "${DECODE_IPS[@]}"; do - echo "[wait] decode http://${_ip}:${DECODE_PORT}/health" - while ! curl -sf --max-time 10 "http://${_ip}:${DECODE_PORT}/health" >/dev/null 2>&1; do - if [[ $(date +%s) -ge $_deadline ]]; then - echo "[wait][FAIL] decode ${_ip}:${DECODE_PORT} not ready after ${WAIT_SERVER_TIMEOUT}s" >&2 - exit 1 - fi - sleep 10 - done - echo "[wait][OK] decode ${_ip}:${DECODE_PORT} ready" - done - fi - echo "[-------]" NODE $NODE_RANK "[--------]" - echo "All servers up. Starting atomesh router..." - - ROUTER_CMD="/usr/local/bin/atomesh launch \ - --host 0.0.0.0 --port ${ROUTER_PORT} \ - --pd-disaggregation \ - ${PREFILL_ARGS} \ - ${DECODE_ARGS} \ - ${ROUTER_POLICY_ARGS} \ - --backend atom \ - --log-level info \ - --disable-health-check \ - --disable-circuit-breaker \ - --prometheus-port 29100" - - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: $ROUTER_CMD" - else - ROUTER_LOG_FILE="/tmp/slurm_job-${SLURM_JOB_ID}_router_${host_name}.log" - set -x - eval "$ROUTER_CMD" 2>&1 | tee "$ROUTER_LOG_FILE" & - set +x - proxy_pid=$! - - check_env_vars WAIT_LOCAL_ROUTER_TIMEOUT - WAIT_ROUTER_TIMEOUT="${WAIT_ROUTER_TIMEOUT:-$WAIT_LOCAL_ROUTER_TIMEOUT}" - echo "[wait] router http://0.0.0.0:${ROUTER_PORT}/v1/models (timeout=${WAIT_ROUTER_TIMEOUT}s)" - _router_deadline=$(( $(date +%s) + WAIT_ROUTER_TIMEOUT )) - while ! curl -sf --max-time 10 "http://0.0.0.0:${ROUTER_PORT}/v1/models" >/dev/null 2>&1; do - if [[ $(date +%s) -ge $_router_deadline ]]; then - echo "[wait][FAIL] router ${ROUTER_PORT}/v1/models not ready after ${WAIT_ROUTER_TIMEOUT}s" >&2 - exit 1 - fi - sleep 10 - done - echo "[wait][OK] router /v1/models ready" - - echo "Router is ready for benchmarking" - fi - - echo "[-------]" NODE $NODE_RANK "[--------]" - echo "Ready for benchmarking on ${host_name}:${host_ip}" - - cd $ATOM_WS_PATH - - export IS_MTP="false" - if [[ -n "$MODEL_MTP_FLAGS" && "${DECODE_MTP_SIZE}" -gt 0 ]]; then - export IS_MTP="true" - fi - - # Select the benchmark runner. - # IS_AGENTIC=1/true -> AgentX trace replay (trace_replay.sh), driven by - # aiperf against the atomesh router on ROUTER_PORT. - # IS_AGENTIC unset/0 -> fixed-seq-len throughput benchmark (bench.sh). - if [[ "$IS_AGENTIC_RUN" == "1" ]]; then - # trace_replay.sh targets ROUTER_PORT and derives MODEL from - # $MODEL_DIR/$MODEL_NAME, which matches the atom server's served-model - # name (its --model path). Each allocation runs one concurrency. - export ROUTER_PORT - export DURATION="${DURATION:-1800}" - # trace_replay.sh / benchmark_lib.sh locate utils/aiperf under - # INFMAX_CONTAINER_WORKSPACE (the container repo root). - # The SGLang client-image path sets it in its env-file; the - # in-container ATOM path must set it too -> derive it from ATOM_WS_PATH - # (.../benchmarks/multi_node/amd_utils -> repo root, i.e. /workspace). - export INFMAX_CONTAINER_WORKSPACE="${INFMAX_CONTAINER_WORKSPACE:-${ATOM_WS_PATH%/benchmarks/multi_node/amd_utils}}" - # trace_replay.sh signature: model_path model_name concurrency_list log_path - BENCH_CMD="bash $ATOM_WS_PATH/trace_replay.sh \ - $MODEL_DIR $MODEL_NAME \"${BENCH_MAX_CONCURRENCY}\" /run_logs/slurm_job-${SLURM_JOB_ID}" - echo "Benchmark runner: trace_replay.sh (agentic ATOM, router :${ROUTER_PORT}, KV_OFFLOADING=${KV_OFFLOADING:-none})" - else - BENCH_CMD="bash $ATOM_WS_PATH/bench.sh ${xP} ${yD} $((PREFILL_TP_SIZE*xP)) $((DECODE_TP_SIZE*yD)) \ - $MODEL_DIR $MODEL_NAME /run_logs/slurm_job-${SLURM_JOB_ID} ${BENCH_INPUT_LEN} \ - ${BENCH_OUTPUT_LEN} \"${BENCH_MAX_CONCURRENCY}\" ${BENCH_REQUEST_RATE} \ - ${BENCH_RANDOM_RANGE_RATIO} ${BENCH_NUM_PROMPTS_MULTIPLIER}" - echo "Benchmark runner: bench.sh (fixed-seq-len)" - fi - - if [[ "${EVAL_ONLY}" == "true" ]]; then - echo "EVAL_ONLY mode: skipping throughput benchmark" - elif [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: $BENCH_CMD" - else - set -x - eval "$BENCH_CMD" - set +x - fi - - if [[ "${RUN_EVAL}" == "true" ]]; then - echo "Running lm-eval evaluation on Node 0..." - - EVAL_HEALTH_OK=false - for _attempt in 1 2 3; do - if curl -sf --max-time 10 "http://0.0.0.0:${ROUTER_PORT}/health" >/dev/null 2>&1; then - EVAL_HEALTH_OK=true - break - fi - echo "Eval health check attempt $_attempt failed, retrying in 10s..." - sleep 10 - done - - if [[ "$EVAL_HEALTH_OK" != "true" ]]; then - echo "WARNING: Router health check failed after 3 attempts. Skipping eval." - else - pushd /workspace - - # job.slurm's -e allowlist forwards ROUTER_PORT but not PORT, and - # run_lm_eval's check_env_vars guard runs before it parses --port. - export PORT="${ROUTER_PORT}" - - source /workspace/benchmarks/benchmark_lib.sh - - if [[ -n "${EVAL_CONC:-}" ]]; then - export EVAL_CONCURRENT_REQUESTS="${EVAL_CONC}" - else - export EVAL_CONCURRENT_REQUESTS=$(echo "$BENCH_MAX_CONCURRENCY" | tr 'x' '\n' | sort -n | tail -1) - fi - - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: run_eval --port ${ROUTER_PORT} (framework=${EVAL_FRAMEWORK}, conc=${EVAL_CONCURRENT_REQUESTS})" - else - MODEL_NAME="${MODEL_DIR}/${MODEL_NAME}" run_eval --port "${ROUTER_PORT}" - eval_rc=$? - - if [[ $eval_rc -ne 0 ]]; then - echo "ERROR: run_eval exited rc=$eval_rc; preserving failure artifacts" >&2 - EVAL_FAILED=1 - else - export TP="${PREFILL_TP_SIZE}" - export CONC="${EVAL_CONCURRENT_REQUESTS}" - export PREFILL_TP="${PREFILL_TP_SIZE}" - export PREFILL_EP=1 - export PREFILL_NUM_WORKERS="${xP}" - export DECODE_TP="${DECODE_TP_SIZE}" - export DECODE_EP=1 - export DECODE_NUM_WORKERS="${yD}" - export ISL="${BENCH_INPUT_LEN}" - export OSL="${BENCH_OUTPUT_LEN}" - - MODEL_NAME="${MODEL_DIR}/${MODEL_NAME}" append_lm_eval_summary - - fi - - EVAL_COPY_DIR="/run_logs/slurm_job-${SLURM_JOB_ID}/eval_results" - if stage_eval_artifacts \ - "$EVAL_COPY_DIR" /workspace "${EVAL_RESULT_DIR:-}"; then - echo "Eval artifacts staged in $EVAL_COPY_DIR" - else - echo "ERROR: failed to stage eval artifacts in $EVAL_COPY_DIR" >&2 - EVAL_FAILED=1 - fi - fi - - popd - fi - fi - - LOGS_OUTPUT="${BENCHMARK_LOGS_DIR}/logs" - mkdir -p "$LOGS_OUTPUT" - if [[ "$DRY_RUN" -eq 0 ]]; then - cp -r /run_logs/slurm_job-${SLURM_JOB_ID} "$LOGS_OUTPUT/" - echo "Copied results to $LOGS_OUTPUT/slurm_job-${SLURM_JOB_ID}" - fi - - echo "Waiting 60s before killing router and prefill server..." - sleep 60 - - echo "[-------]" NODE $NODE_RANK "[--------]" - echo "Killing router and prefill server" - if [[ "$DRY_RUN" -eq 0 ]]; then - kill $proxy_pid - kill $prefill0_pid - fi - - if [[ "${EVAL_FAILED:-0}" -eq 1 ]]; then - echo "ERROR: eval failed; exiting node-0 with rc=1" - exit 1 - fi - -elif [ "$NODE_RANK" -gt 0 ] && [ "$NODE_RANK" -lt "$NODE_OFFSET" ]; then - echo "${host_name}:${host_ip} is Prefill Node (rank ${NODE_RANK})" - - prefill_worker_idx=$((NODE_RANK / PREFILL_NODES_PER_WORKER)) - PREFILL_HEADNODE_IP="${PREFILL_IPS[$prefill_worker_idx]}" - - PREFILL_CMD="python3 -m atom.entrypoints.openai_server \ - --model ${MODEL_DIR}/${MODEL_NAME} \ - --host 0.0.0.0 --server-port ${PREFILL_PORT} \ - --trust-remote-code \ - ${PREFILL_PARALLEL_ARGS[*]} \ - ${SPEC_ARGS[*]} \ - ${KV_CACHE_ARG} \ - --block-size ${BLOCK_SIZE} \ - --gpu-memory-utilization ${MEM_FRAC_STATIC} \ - --max-num-seqs ${MAX_NUM_SEQS} \ - ${MODEL_LEN_ARGS} \ - ${AGENTIC_SERVER_ARGS} \ - ${PREFIX_CACHE_ARG} \ - ${ONLINE_QUANT_ARG} \ - --kv-transfer-config '${PREFILL_KV_TRANSFER}' \ - ${EXTRA_SERVER_ARGS}" - - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: $PREFILL_CMD" - else - set -x - eval "$PREFILL_CMD" \ - 2>&1 | tee /run_logs/slurm_job-${SLURM_JOB_ID}/prefill_${host_name}.log & - set +x - prefill_pid=$! - trap 'echo "Caught signal, killing prefill (pid=$prefill_pid)"; kill $prefill_pid 2>/dev/null; exit 0' SIGTERM SIGINT - fi - - echo "[-------]" NODE $NODE_RANK "[--------]" - echo "Waiting for router to be up..." - check_env_vars WAIT_REMOTE_ROUTER_TIMEOUT - WAIT_ROUTER_TIMEOUT="${WAIT_ROUTER_TIMEOUT:-$WAIT_REMOTE_ROUTER_TIMEOUT}" - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: wait for router ${NODE0_ADDR}:${ROUTER_PORT}/health" - else - _router_deadline=$(( $(date +%s) + WAIT_ROUTER_TIMEOUT )) - while ! curl -sf --max-time 10 "http://${NODE0_ADDR}:${ROUTER_PORT}/health" >/dev/null 2>&1; do - if [[ $(date +%s) -ge $_router_deadline ]]; then - echo "[wait][FAIL] router ${NODE0_ADDR}:${ROUTER_PORT} not ready after ${WAIT_ROUTER_TIMEOUT}s" >&2 - exit 1 - fi - sleep 10 - done - echo "[wait][OK] router ${NODE0_ADDR}:${ROUTER_PORT} ready" - fi - - echo "[-------]" NODE $NODE_RANK "[--------]" - echo "Waiting until router closes..." - trap 'echo "Caught signal, killing prefill (pid=$prefill_pid)"; kill $prefill_pid 2>/dev/null; exit 0' SIGTERM SIGINT - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: wait until router ${NODE0_ADDR}:${ROUTER_PORT} closes" - else - while curl -sf --max-time 10 "http://${NODE0_ADDR}:${ROUTER_PORT}/health" >/dev/null 2>&1; do - sleep 10 & - wait $! - done - echo "[wait] router ${NODE0_ADDR}:${ROUTER_PORT} closed" - fi - - echo "[-------]" NODE $NODE_RANK "[--------]" - echo "Killing prefill server (rank ${NODE_RANK})" - if [[ "$DRY_RUN" -eq 0 ]]; then kill $prefill_pid 2>/dev/null; fi - -else - RANK=$((NODE_RANK - NODE_OFFSET)) - echo "${host_name}:${host_ip} is Decode Node (rank ${RANK})" - - _MAX_CONC=$(echo "$BENCH_MAX_CONCURRENCY" | tr 'x' '\n' | sort -n | tail -1) - CUDAGRAPH_SIZES='[1,2,4,8,16,24,32,40,48,56,64,72,80,88,96,104,112,120,128,136,144,152,160,168,176,184,192,200,208,216,224,232,240,248,256]' - - if [[ "$IS_AGENTIC_RUN" == "1" ]]; then - # Recipe max-num-seqs is 2*concurrency. - DECODE_MAX_NUM_SEQS=$((2 * _MAX_CONC)) - # Dense capture ladder per the recipe's per-tier decode sizing: - # TP decode (no DP attention): 1..min(64, 2*conc). - # DP-attention decode: per-rank 1..(conc/4), since max-num-seqs=2*conc - # spreads across the 8 DP ranks (2*conc / 8 = conc/4). - # Every batch size up to the cap gets a graph, which measurably helps - # small-batch agentic decode. - if [[ "$DECODE_ENABLE_DP" == "true" ]]; then - _dense_max=$((_MAX_CONC / 4)) - else - _dense_max=$((2 * _MAX_CONC)) - if [[ "$_dense_max" -gt 64 ]]; then _dense_max=64; fi - fi - if [[ "$_dense_max" -lt 1 ]]; then _dense_max=1; fi - CUDAGRAPH_SIZES="[$(seq -s, 1 "$_dense_max")]" - else - DECODE_MAX_NUM_SEQS="${_MAX_CONC}" - fi - - DECODE_CMD="python3 -m atom.entrypoints.openai_server \ - --model ${MODEL_DIR}/${MODEL_NAME} \ - --host 0.0.0.0 --server-port ${DECODE_PORT} \ - --trust-remote-code \ - ${DECODE_PARALLEL_ARGS[*]} \ - ${SPEC_ARGS[*]} \ - ${KV_CACHE_ARG} \ - --block-size ${BLOCK_SIZE} \ - --gpu-memory-utilization ${MEM_FRAC_STATIC} \ - --max-num-seqs ${DECODE_MAX_NUM_SEQS} \ - ${MODEL_LEN_ARGS} \ - ${AGENTIC_SERVER_ARGS} \ - ${PREFIX_CACHE_ARG} \ - ${ONLINE_QUANT_ARG} \ - --kv-transfer-config '${DECODE_KV_TRANSFER}' \ - --cudagraph-capture-sizes "${CUDAGRAPH_SIZES}" \ - ${EXTRA_SERVER_ARGS}" - - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: $DECODE_CMD" - else - set -x - eval "$DECODE_CMD" \ - 2>&1 | tee /run_logs/slurm_job-${SLURM_JOB_ID}/decode_${host_name}.log & - set +x - decode_pid=$! - trap 'echo "Caught signal, killing decode (pid=$decode_pid)"; kill $decode_pid 2>/dev/null; exit 0' SIGTERM SIGINT - fi - - echo "[-------]" NODE $NODE_RANK "[--------]" - echo "Waiting for router to be up..." - check_env_vars WAIT_REMOTE_ROUTER_TIMEOUT - WAIT_ROUTER_TIMEOUT="${WAIT_ROUTER_TIMEOUT:-$WAIT_REMOTE_ROUTER_TIMEOUT}" - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: wait for router ${NODE0_ADDR}:${ROUTER_PORT}/health" - else - _router_deadline=$(( $(date +%s) + WAIT_ROUTER_TIMEOUT )) - while ! curl -sf --max-time 10 "http://${NODE0_ADDR}:${ROUTER_PORT}/health" >/dev/null 2>&1; do - if [[ $(date +%s) -ge $_router_deadline ]]; then - echo "[wait][FAIL] router ${NODE0_ADDR}:${ROUTER_PORT} not ready after ${WAIT_ROUTER_TIMEOUT}s" >&2 - exit 1 - fi - sleep 10 - done - echo "[wait][OK] router ${NODE0_ADDR}:${ROUTER_PORT} ready" - fi - - echo "[-------]" NODE $NODE_RANK "[--------]" - echo "Waiting until router closes..." - trap 'echo "Caught signal, killing decode (pid=$decode_pid)"; kill $decode_pid 2>/dev/null; exit 0' SIGTERM SIGINT - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: wait until router ${NODE0_ADDR}:${ROUTER_PORT} closes" - else - while curl -sf --max-time 10 "http://${NODE0_ADDR}:${ROUTER_PORT}/health" >/dev/null 2>&1; do - sleep 10 & - wait $! - done - echo "[wait] router ${NODE0_ADDR}:${ROUTER_PORT} closed" - fi - - echo "[-------]" NODE $NODE_RANK "[--------]" - echo "Killing decode server (rank ${RANK})" - if [[ "$DRY_RUN" -eq 0 ]]; then kill $decode_pid 2>/dev/null; fi -fi - -echo "Script completed successfully" -exit 0 diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/atom/mi355x-fp4/agentx/disagg-lmcache-dspark.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/atom/mi355x-fp4/agentx/disagg-lmcache-dspark.yaml new file mode 100644 index 0000000000..6c960f5b01 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/atom/mi355x-fp4/agentx/disagg-lmcache-dspark.yaml @@ -0,0 +1,254 @@ +# DeepSeek-V4-Pro-0813 AgentX on MI355X: 1P1D ATOM disaggregation over Mooncake +# RDMA behind AToMesh, with DSpark (three draft tokens). Ported from the legacy +# amd_utils path (#3158; ATOM recipes/DeepSeek-V4-Agentic-PD-Max.md). Three tiers: +# TP8 at concurrency 1-16, DP attention at 64-128, and DP attention plus ATOM's +# in-process LMCache CPU offload on prefill at 256. srtctl generates the Mooncake +# P/D connector; the offload tier adds lmcache_offload next to it through +# extra-kv-connectors. +base: + schema: 2 + name: mi355x-dsv4-pro-0813-atom-agentx-lmcache + model: + path: DeepSeek-V4-Pro-0813 + container: rocm/atom-dev:nightly_202609221542 + precision: fp4 + slurm: + time_limit: "08:00:00" + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: atomesh + enable_multiple_frontends: false + args: + log-level: info + disable-health-check: true + disable-circuit-breaker: true + prometheus-port: 29100 + engine: atom + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 8 + env: &environment + PYTHONUNBUFFERED: "1" + PYTHONDONTWRITEBYTECODE: "1" + SAFETENSORS_FAST_GPU: "1" + VLLM_LOG_LEVEL: WARNING + ATOM_LOG_LEVEL: WARNING + AITER_LOG_LEVEL: WARNING + LOG_LEVEL: WARNING + LOGLEVEL: WARNING + ATOM_MOE_GU_ITLV: "1" + AITER_BF16_FP8_MOE_BOUND: "0" + # Rail-isolated RDMA: matched rails replace dest-device affinity. + MC_GID_INDEX: "1" + ATOM_MOONCAKE_MATCHED_RAILS: auto + NCCL_IB_DISABLE: "1" + ATOM_DISABLE_MMAP: "true" + ATOM_NUMA_BIND: "1" + ATOM_DP_MASTER_PORT: "29510" + ATOM_DP_BASE_PORT: "29610" + ATOM_PREFIX_CACHE_POLICY: lru + ATOM_PREFIX_CACHE_PROTECTED_RATIO: "0.5" + args: &server + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + trust-remote-code: true + method: dspark + num-speculative-tokens: 3 + kv_cache_dtype: fp8 + index-cache-dtype: fp4 + # ATOM forces block size 256 for V4. + block-size: 256 + gpu-memory-utilization: 0.9 + max-model-len: 1000000 + max-num-batched-tokens: 16384 + attn-prefill-chunk-size: 16384 + state-checkpoint-interval-tokens: 8192 + level: 3 + cudagraph-mode: FULL + enable-prefix-caching: true + decode: + nodes: 1 + workers: 1 + gpus: 8 + env: + <<: *environment + args: + <<: *server + sbatch_directives: + cpus-per-task: "128" + mem: "0" + srun_options: + mem: "0" + container-writable: "" + container-remap-root: "" + health_check: + max_attempts: 720 + interval_seconds: 5 + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/srt_agentic.sh + env: + MODEL: deepseek-ai/DeepSeek-V4-Pro-0813 + # The MI355X launcher collects results from the job workspace. + RESULT_DIR: /infmax-workspace/LOGS/agentic + AGENTIC_OUTPUT_DIR: /infmax-workspace + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "atom:" + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + HF_HUB_CACHE: /hf_hub_cache/hub + TOKENIZERS_PARALLELISM: "false" + TRANSFORMERS_VERBOSITY: error + +# One variant per point; each point runs in its own allocation. Admission is +# 2x CONC on both roles. Decode captures every batch up to min(64, 2x CONC) on +# TP, or CONC/4 per rank under DP attention. DP attention adds TBO on prefill. + +override_tp8_c1: + frontend: + args: + prefill-policy: round_robin + decode-policy: round_robin + atom-pd-rank-mapping-policy: none + roles: + prefill: + args: + max-num-seqs: 2 + decode: + args: + max-num-seqs: 2 + cudagraph-capture-sizes: '[1,2]' + benchmark: + env: + CONC: '1' + +override_tp8_c16: + frontend: + args: + prefill-policy: round_robin + decode-policy: round_robin + atom-pd-rank-mapping-policy: none + roles: + prefill: + args: + max-num-seqs: 32 + decode: + args: + max-num-seqs: 32 + cudagraph-capture-sizes: '[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32]' + benchmark: + env: + CONC: '16' + +override_dpa8_c64: + frontend: + args: + dp-aware: true + prefill-policy: cache_aware + decode-policy: cache_aware + cache-threshold: 0.8 + balance-abs-threshold: 20 + balance-rel-threshold: 2.0 + eviction-interval: 300 + atom-pd-rank-mapping-policy: none + roles: + prefill: + env: + ATOM_ENABLE_PREFILL_DELAYER: '0' + GPU_MAX_HW_QUEUES: '5' + args: + max-num-seqs: 128 + enable-dp-attention: true + enable-tbo: true + decode: + env: + ATOM_ENABLE_PREFILL_DELAYER: '0' + args: + max-num-seqs: 128 + cudagraph-capture-sizes: '[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16]' + enable-dp-attention: true + benchmark: + env: + CONC: '64' + +override_dpa8_c128: + frontend: + args: + dp-aware: true + prefill-policy: cache_aware + decode-policy: cache_aware + cache-threshold: 0.8 + balance-abs-threshold: 20 + balance-rel-threshold: 2.0 + eviction-interval: 300 + atom-pd-rank-mapping-policy: none + roles: + prefill: + env: + ATOM_ENABLE_PREFILL_DELAYER: '0' + GPU_MAX_HW_QUEUES: '5' + args: + max-num-seqs: 256 + enable-dp-attention: true + enable-tbo: true + decode: + env: + ATOM_ENABLE_PREFILL_DELAYER: '0' + args: + max-num-seqs: 256 + cudagraph-capture-sizes: '[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32]' + enable-dp-attention: true + benchmark: + env: + CONC: '128' + +override_dpa8_c256_lmcache: + frontend: + args: + dp-aware: true + prefill-policy: dp_sticky + decode-policy: dp_sticky + atom-pd-rank-mapping-policy: none + roles: + prefill: + env: + ATOM_ENABLE_PREFILL_DELAYER: '0' + GPU_MAX_HW_QUEUES: '5' + PYTHONHASHSEED: '0' + OFFLOAD_COPY_WORKERS: '1' + OFFLOAD_MIN_LOAD_TOKENS: '8192' + OFFLOAD_SLOT_STAGING_SLOTS: '4' + args: + max-num-seqs: 512 + enable-dp-attention: true + enable-tbo: true + extra-kv-connectors: + - kv_connector: lmcache_offload + kv_role: offload + offload_layout: hybrid + max_pending_saves: 8 + slot_sidecar_staging_slots: 4 + lmcache.local_cpu: true + # total-cpu-dram-gb 1499 (dram-utilization 0.5) over the eight prefill ranks. + lmcache.max_local_cpu_size: 187 + lmcache.local_disk: null + lmcache.max_local_disk_size: 0 + lmcache.remote_url: null + lmcache.chunk_size: 256 + lmcache.cache_policy: LRU + lmcache.lookup_server_worker_ids: [] + lmcache.store_location: LocalCPUBackend + lmcache.retrieve_locations: [LocalCPUBackend] + decode: + env: + ATOM_ENABLE_PREFILL_DELAYER: '0' + args: + max-num-seqs: 512 + cudagraph-capture-sizes: '[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64]' + enable-dp-attention: true + benchmark: + env: + CONC: '256' + # dp_sticky keys sessions on the correlation id. + AIPERF_HTTP_X_SESSION_ID_FROM_CORRELATION_ID: '1' diff --git a/inferencex-e2e/configs/amd-master.yaml b/inferencex-e2e/configs/amd-master.yaml index 62b370ca5d..50b957ff30 100644 --- a/inferencex-e2e/configs/amd-master.yaml +++ b/inferencex-e2e/configs/amd-master.yaml @@ -1266,13 +1266,13 @@ dsv4-fp4-mi355x-atom-disagg-agentic-lmcache-dspark: disagg: true scenarios: agentic-coding: - # MTP tiers with GPU-resident KV, unchanged from run 35643358891 apart from - # the image bump. Concurrency 1/16 run plain TP8; 64/128 add DP attention - # (c128 measured in run 35810485934, gsm8k eval in run 35851118405). + # srt-slurm recipe benchmarks/multi_node/srt-slurm-recipes/dsv4/atom/mi355x-fp4/agentx/disagg-lmcache-dspark.yaml + # (ATOM recipes/DeepSeek-V4-Agentic-PD-Max.md). Concurrency 1/16 run plain + # TP8; 64/128 add DP attention with prefill TBO and the cache-aware router. - dram-utilization: 0.80 search-space: - spec-decoding: "draft_model" - conc-list: [ 1, 16 ] + conc-list: [ 1 ] kv-offloading: none prefill: num-worker: 1 @@ -1280,19 +1280,29 @@ dsv4-fp4-mi355x-atom-disagg-agentic-lmcache-dspark: ep: 1 dp-attn: false additional-settings: - - "PREFILL_NODES=1" + - "CONFIG_FILE=recipes/dsv4/atom/mi355x-fp4/agentx/disagg-lmcache-dspark.yaml:override_tp8_c1" decode: num-worker: 1 tp: 8 ep: 1 dp-attn: false + - spec-decoding: "draft_model" + conc-list: [ 16 ] + kv-offloading: none + prefill: + num-worker: 1 + tp: 8 + ep: 1 + dp-attn: false additional-settings: - - "DECODE_NODES=1" - - "DECODE_MTP_SIZE=3" - # DP-attention tier (recipe: concurrency 64-128). TP8 + DP attention, - # prefill TBO only, cache-aware router, no CPU offload. + - "CONFIG_FILE=recipes/dsv4/atom/mi355x-fp4/agentx/disagg-lmcache-dspark.yaml:override_tp8_c16" + decode: + num-worker: 1 + tp: 8 + ep: 1 + dp-attn: false - spec-decoding: "draft_model" - conc-list: [ 64, 128 ] + conc-list: [ 64 ] kv-offloading: none prefill: num-worker: 1 @@ -1300,18 +1310,30 @@ dsv4-fp4-mi355x-atom-disagg-agentic-lmcache-dspark: ep: 1 dp-attn: true additional-settings: - - "PREFILL_NODES=1" + - "CONFIG_FILE=recipes/dsv4/atom/mi355x-fp4/agentx/disagg-lmcache-dspark.yaml:override_dpa8_c64" decode: num-worker: 1 tp: 8 ep: 1 dp-attn: true + - spec-decoding: "draft_model" + conc-list: [ 128 ] + kv-offloading: none + prefill: + num-worker: 1 + tp: 8 + ep: 1 + dp-attn: true additional-settings: - - "DECODE_NODES=1" - - "DECODE_MTP_SIZE=3" - # DP-attention + CPU offload tier (recipe: concurrency 256). Prefill adds - # the lmcache host KV offload tier; decode is plain Mooncake. DSpark draft - # model and dram-utilization 0.50 as measured in run 35825955863. + - "CONFIG_FILE=recipes/dsv4/atom/mi355x-fp4/agentx/disagg-lmcache-dspark.yaml:override_dpa8_c128" + decode: + num-worker: 1 + tp: 8 + ep: 1 + dp-attn: true + # DP attention plus ATOM's in-process LMCache CPU offload on prefill + # (lmcache_offload); decode is plain Mooncake. dram-utilization 0.50 as + # measured in run 35825955863. - dram-utilization: 0.50 search-space: - spec-decoding: "draft_model" @@ -1324,15 +1346,12 @@ dsv4-fp4-mi355x-atom-disagg-agentic-lmcache-dspark: ep: 1 dp-attn: true additional-settings: - - "PREFILL_NODES=1" + - "CONFIG_FILE=recipes/dsv4/atom/mi355x-fp4/agentx/disagg-lmcache-dspark.yaml:override_dpa8_c256_lmcache" decode: num-worker: 1 tp: 8 ep: 1 dp-attn: true - additional-settings: - - "DECODE_NODES=1" - - "DECODE_MTP_SIZE=3" # GLM-5.2 FP4 agentic-coding benchmark on MI355X via SGLang with MTP speculative # decoding. Two arms: diff --git a/inferencex-e2e/infx/launch/drivers/srt/submit.py b/inferencex-e2e/infx/launch/drivers/srt/submit.py index dce2daf283..a6319192f2 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/submit.py +++ b/inferencex-e2e/infx/launch/drivers/srt/submit.py @@ -89,12 +89,15 @@ def apply( ) -> subprocess.CompletedProcess[str]: """Run ``srtctl apply`` for ``config``, through the golden AgentX acceptance planner. - ``stdout`` receives srtctl's JSON manifest; without it stdout and stderr are - captured and echoed. + Every container starts in the workspace mount, as the legacy launchers did: + PyTorch's generated module imports fail from / with PYTHONPYCACHEPREFIX set. + ``arguments`` still win. ``stdout`` receives srtctl's JSON manifest; without + it stdout and stderr are captured and echoed. """ argv = [ str(checkout.venv / "bin/python"), "-m", "infx.srt_slurm.synthetic_acceptance", - config, run.request.framework, "--", *arguments, + config, run.request.framework, "--", + "--set", 'srun_options.container-workdir="/infmax-workspace"', *arguments, ] # fmt: skip env = {**run.env, "RUNNER_NAME": srtctl_job_name(run.request.runner_name)} if stdout is None: diff --git a/inferencex-e2e/infx/srt_slurm/single_node.py b/inferencex-e2e/infx/srt_slurm/single_node.py index 0c758d1f45..136b1afe51 100644 --- a/inferencex-e2e/infx/srt_slurm/single_node.py +++ b/inferencex-e2e/infx/srt_slurm/single_node.py @@ -145,9 +145,6 @@ def runtime_arguments(config: str, environment: Mapping[str, str]) -> list[str]: # Exclusive nodes include idle GPUs. Restrict each server/client step to # the serving GPU count so client-side power collection sees the same set. overrides = ["--set", f"srun_options.gpus-per-node={json.dumps(environment['GPU_COUNT'])}"] - # Match the legacy container working directory using the existing repo mount. - # PyTorch's generated module imports fail from / with PYTHONPYCACHEPREFIX set. - overrides += ["--set", 'srun_options.container-workdir="/infmax-workspace"'] if environment.get("SRT_SRUN_OPTIONS"): options = json.loads(environment["SRT_SRUN_OPTIONS"]) if not isinstance(options, dict) or any( diff --git a/inferencex-e2e/infx/srt_slurm/synthetic_acceptance.py b/inferencex-e2e/infx/srt_slurm/synthetic_acceptance.py index 45a9f2e153..691c86c009 100644 --- a/inferencex-e2e/infx/srt_slurm/synthetic_acceptance.py +++ b/inferencex-e2e/infx/srt_slurm/synthetic_acceptance.py @@ -28,6 +28,7 @@ "trt": "trtllm", "dynamo-trt": "trtllm", "atom": "atom", + "atom-disagg": "atom", } SGLANG_VARIABLES = ( "SGLANG_SIMULATE_ACC_LEN", diff --git a/inferencex-e2e/infx/tests/launch/test_srt_driver.py b/inferencex-e2e/infx/tests/launch/test_srt_driver.py index da513d1dd4..1377d2beac 100644 --- a/inferencex-e2e/infx/tests/launch/test_srt_driver.py +++ b/inferencex-e2e/infx/tests/launch/test_srt_driver.py @@ -124,6 +124,9 @@ def srtslurm(root: Path) -> dict: return yaml.safe_load(path.read_text()) +WORKDIR = 'srun_options.container-workdir="/infmax-workspace"' + + def assert_ok(result: subprocess.CompletedProcess[str]) -> None: """Fail with the launcher's output when it did not exit 0.""" assert result.returncode == 0, result.stdout[-4000:] + result.stderr[-8000:] @@ -142,7 +145,7 @@ def test_single_node_point_stages_workflow_artifacts(harness): [call] = srtctl_calls(harness.logs) argv = call["argv"] assert argv[argv.index("--file") + 1] == f"{workspace}/recipe.yaml:zip_override_conc[0]" - assert {"--json", "--yes", "--output"} <= set(argv) + assert {"--json", "--yes", "--output", WORKDIR} <= set(argv) assert (call["env"]["INFMAX_WORKSPACE"], call["env"]["VIRTUAL_ENV"]) == (str(workspace), None) assert call["env"]["RUNNER_NAME"] == f"inferencex-{env['RUNNER_NAME']}" assert "/hf" in srtslurm(workspace)["default_mounts"].values() @@ -256,7 +259,7 @@ def test_multinode_lane_stages_workflow_artifacts(harness, monkeypatch, cluster_ assert checkout.name.startswith("srt-slurm-9001-1-") and len(checkout.name) == len("srt-slurm-9001-1-") + 12 assert ("--no-preflight" in argv) is not lab["preflight"] assert argv[argv.index("--file") + 1] == "recipes/test/lane.yaml" - assert {"--json", "--yes", "benchmark.stream_output=true"} <= set(argv) + assert {"--json", "--yes", "benchmark.stream_output=true", WORKDIR} <= set(argv) if lab["tag"] is None: assert "--tags" not in argv else: diff --git a/inferencex-e2e/infx/tests/srt_slurm/test_srt_single_node.py b/inferencex-e2e/infx/tests/srt_slurm/test_srt_single_node.py index 7df3de526f..f7463515d8 100644 --- a/inferencex-e2e/infx/tests/srt_slurm/test_srt_single_node.py +++ b/inferencex-e2e/infx/tests/srt_slurm/test_srt_single_node.py @@ -52,9 +52,7 @@ def test_native_binding_submits_one_point_and_keeps_server_settings(point): overrides = parse_overrides(argv[1::2], []) actual = copy.deepcopy(recipe) apply_overrides_to_recipe(actual, overrides) - assert actual["srun_options"] == { - "gpus-per-node": "4", "container-workdir": "/infmax-workspace", - } + assert actual["srun_options"] == {"gpus-per-node": "4"} assert actual["benchmark"]["env"] == { "MODEL": "test/model", "ISL": "256", "OSL": "64", "RANDOM_RANGE_RATIO": "0.5", "USE_CHAT_TEMPLATE": "false", diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index d7fc64f0d6..b415cf56a5 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9149,3 +9149,12 @@ - "Use AITER attention and allreduce fusion, FP8 KV (fp8_e4m3), and EAGLE MTP (3 steps, topk 1, 4 draft tokens); do not enable ROCm INT4 quick all-reduce. HiCache uses ratio 1.5, write_through_selective, kernel I/O and page_first layout. The matching cookbook recipe is sgl-project/sglang#41849." - "The embedded MTP head runs at its stored precision: block-FP8 expert weights and checkpoint dtype for mtp.fc and gates. No separate draft, draft dtype override or submission-side quantization is used." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3602 + +- config-keys: + - dsv4-fp4-mi355x-atom-disagg-agentic-lmcache-dspark + scenario-type: + - agentic-coding + description: + - "Port the DeepSeek-V4-Pro-0813 MI355X ATOM 1P1D AgentX config from the legacy amd_utils path to a native srt-slurm recipe (benchmarks/multi_node/srt-slurm-recipes/dsv4/atom/mi355x-fp4/agentx/disagg-lmcache-dspark.yaml), one override variant per point. Server flags, env, AToMesh routing policies and decode CUDA-graph capture sizes match the legacy server_atom.sh for every tier: TP8 at concurrency 1 and 16, DP attention with prefill TBO at 64 and 128, and DP attention with ATOM's in-process LMCache CPU offload (lmcache_offload, 187 GB per prefill rank) at 256. Golden acceptance 3.01 for DSpark with three draft tokens is now injected by the srt-slurm path, the same value the legacy models_atom.yaml hardcoded." + - "The LMCache tier uses srt-slurm's extra-kv-connectors (NVIDIA/srt-slurm#507, in v2.36.0) to add lmcache_offload next to the generated Mooncake connector." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3543 diff --git a/inferencex-e2e/runners/srt-slurm/patches/548-atom-served-model-name.patch b/inferencex-e2e/runners/srt-slurm/patches/548-atom-served-model-name.patch new file mode 100644 index 0000000000..7aaa73673a --- /dev/null +++ b/inferencex-e2e/runners/srt-slurm/patches/548-atom-served-model-name.patch @@ -0,0 +1,54 @@ +diff --git a/src/srtctl/backends/atom.py b/src/srtctl/backends/atom.py +index 8aeea3a0..60659222 100644 +--- a/src/srtctl/backends/atom.py ++++ b/src/srtctl/backends/atom.py +@@ -82,6 +82,13 @@ class AtomProtocol: + return {} + + def get_served_model_name(self, default: str) -> str: ++ """The name ATOM serves: a role's ``served-model-name``, else its literal ``--model``.""" ++ if self.atom_config: ++ for cfg in [self.atom_config.prefill, self.atom_config.aggregated, self.atom_config.decode]: ++ if cfg: ++ name = cfg.get("served-model-name") or cfg.get("served_model_name") ++ if name: ++ return name + return default + + @property +diff --git a/src/srtctl/core/schema.py b/src/srtctl/core/schema.py +index 16edddaf..915fcbc5 100755 +--- a/src/srtctl/core/schema.py ++++ b/src/srtctl/core/schema.py +@@ -3369,9 +3369,9 @@ class SrtConfig: + role = get_frontend(self.frontend.type).model_name_role or role + backend = self.backend_for_role(role) + if isinstance(backend, AtomProtocol): +- # ATOM advertises the literal --model argument; unlike SGLang/vLLM, +- # it has no separate served-model-name alias. Match the worker's +- # HF ID or container-visible path, including node-local staging. ++ # Without served-model-name, ATOM advertises the literal --model ++ # argument: the worker's HF ID or container-visible path, including ++ # node-local staging. + model_path = os.path.expandvars(self.model.path) + if model_path.startswith("hf:"): + default = model_path[3:] +diff --git a/tests/test_atom_atomesh.py b/tests/test_atom_atomesh.py +index 7f4d017f..7f68cd39 100644 +--- a/tests/test_atom_atomesh.py ++++ b/tests/test_atom_atomesh.py +@@ -133,6 +133,14 @@ def test_atom_served_name_matches_worker_model_argument(tmp_path: Path, layout: + assert config.served_model_name == expected + + ++def test_atom_served_name_follows_served_model_name() -> None: ++ """A role's served-model-name is the name ATOM and AToMesh serve, so evals must send it.""" ++ data = _config() ++ for role in ("prefill", "decode"): ++ data["roles"][role].setdefault("args", {})["served-model-name"] = "qwen3-served" ++ assert _load(data).served_model_name == "qwen3-served" ++ ++ + def test_atom_builds_native_aggregate_command() -> None: + """Recipe flags keep ATOM's mixed hyphen/underscore spelling and follow the managed arguments.""" + backend = AtomProtocol( diff --git a/inferencex-e2e/runners/srt-slurm/patches/README.md b/inferencex-e2e/runners/srt-slurm/patches/README.md index ffc4c1557a..ba9db355f1 100644 --- a/inferencex-e2e/runners/srt-slurm/patches/README.md +++ b/inferencex-e2e/runners/srt-slurm/patches/README.md @@ -8,3 +8,4 @@ Each patch is a temporary fix for an open upstream PR. When the PR merges and th | Patch | Upstream PR | Fix | |-------|-------------|-----| +| `548-atom-served-model-name.patch` | [NVIDIA/srt-slurm#548](https://github.com/NVIDIA/srt-slurm/pull/548) | ATOM serves evals the role's `served-model-name` instead of the literal `--model` path, so AToMesh accepts eval requests |