From 88ad6a294a1b4df30ce1045a02584294f73f713b Mon Sep 17 00:00:00 2001 From: "baijie.xie" Date: Sun, 9 Aug 2026 00:04:05 +0800 Subject: [PATCH] [tilert] Add glm5.1-fp8-b200-tilert 1k1k+8k1k (prefill/decode disagg via TileRT) In-repo duplicate of https://github.com/SemiAnalysisAI/InferenceX/pull/2523 by @CrimsonDump, so the sweep can run on run-sweep.yml and merge through the /reuse-sweep-run path (fork PRs are limited to the trusted external e2e dispatch, which is not reuse-eligible). Content is the fork branch squashed at d5f23869: the original recipe, the staged-weights + RoCE device-list fix, the NVIDIA_VISIBLE_DEVICES export that enroot needs to inject the driver into the tilert image, and the perf-changelog union-merge with main. Validated end-to-end green on the trusted external run 31260389587 (b200-dgxc, both multi-node legs). --- .../glm5.1_fp8_b200_tilert-disagg.sh | 34 +++ .../multi_node/tilert_utils/run_node.sh | 263 ++++++++++++++++++ .../multi_node/tilert_utils/setup_deps.sh | 142 ++++++++++ benchmarks/multi_node/tilert_utils/submit.sh | 74 +++++ configs/nvidia-master.yaml | 55 ++++ perf-changelog.yaml | 9 + runners/launch_b200-dgxc.sh | 23 ++ 7 files changed, 600 insertions(+) create mode 100755 benchmarks/multi_node/glm5.1_fp8_b200_tilert-disagg.sh create mode 100755 benchmarks/multi_node/tilert_utils/run_node.sh create mode 100644 benchmarks/multi_node/tilert_utils/setup_deps.sh create mode 100755 benchmarks/multi_node/tilert_utils/submit.sh diff --git a/benchmarks/multi_node/glm5.1_fp8_b200_tilert-disagg.sh b/benchmarks/multi_node/glm5.1_fp8_b200_tilert-disagg.sh new file mode 100755 index 0000000000..b7fffba9fc --- /dev/null +++ b/benchmarks/multi_node/glm5.1_fp8_b200_tilert-disagg.sh @@ -0,0 +1,34 @@ +#!/usr/bin/env bash + +source "$(dirname "$0")/../benchmark_lib.sh" + +check_env_vars \ + CONC_LIST \ + ISL \ + OSL \ + IMAGE \ + SPEC_DECODING \ + MODEL_PATH \ + PREFILL_NUM_WORKERS \ + PREFILL_TP \ + PREFILL_EP \ + PREFILL_DP_ATTN \ + DECODE_NUM_WORKERS \ + DECODE_TP \ + DECODE_EP \ + DECODE_DP_ATTN \ + PREFILL_NODES \ + DECODE_NODES \ + RANDOM_RANGE_RATIO \ + FRAMEWORK + +export MODEL_NAME=glm5 +export TILERT_MODEL_TYPE=glm-5 +export MAX_MODEL_LEN="${MAX_MODEL_LEN:-202752}" + +export DECODE_KV_DTYPE=fp8 +export PREFILL_KV_DTYPE=fp8_ds_mla + +export TILERT_PARSER=none + +exec bash "$(dirname "$0")/tilert_utils/submit.sh" diff --git a/benchmarks/multi_node/tilert_utils/run_node.sh b/benchmarks/multi_node/tilert_utils/run_node.sh new file mode 100755 index 0000000000..c6d3f0046c --- /dev/null +++ b/benchmarks/multi_node/tilert_utils/run_node.sh @@ -0,0 +1,263 @@ +#!/usr/bin/env bash + +source "$(dirname "$0")/../../benchmark_lib.sh" + +MODEL_NAME=${MODEL_NAME:-glm5} +MAX_MODEL_LEN=${MAX_MODEL_LEN:-202752} +GPU_MEM_UTIL=${GPU_MEM_UTIL:-0.75} +RESULT_DIR=${RESULT_DIR:-/workspace} +BENCHMARK_LOGS_DIR=${BENCHMARK_LOGS_DIR:-/workspace} + +DECODE_CTRL_PORT=${DECODE_CTRL_PORT:-5556} +DECODE_HTTP_PORT=${DECODE_HTTP_PORT:-5557} +PREFILL_PORT=${PREFILL_PORT:-8000} +ROUTER_PORT=${PORT:-8888} +DECODE_WAIT=${DECODE_WAIT:-3600} +KV_P2P_TRANSFER=${KV_P2P_TRANSFER:-nixl} +TILERT_WEIGHTS_DIR=${TILERT_WEIGHTS_DIR:-/workspace/GLM-5-FP8-TileRT} +TILERT_MODEL_TYPE=${TILERT_MODEL_TYPE:-glm-5} +TILERT_PARSER=${TILERT_PARSER:-none} +DECODE_KV_DTYPE=${DECODE_KV_DTYPE:-fp8} +PREFILL_KV_DTYPE=${PREFILL_KV_DTYPE:-fp8_ds_mla} + +PREFILL_SPEC=(--speculative-config '{"method":"mtp","num_speculative_tokens":1}') +DECODE_MTP=(--with-mtp) + +: "${DECODE_HOST:?DECODE_HOST is unset -- submit.sh must export it}" +: "${PREFILL_HOST:?PREFILL_HOST is unset -- submit.sh must export it}" +: "${TILERT_ROLE:?TILERT_ROLE is unset -- submit.sh must set it to decode or prefill}" + +mkdir -p "$BENCHMARK_LOGS_DIR" + +DONE_SENTINEL="$BENCHMARK_LOGS_DIR/.tilert_done.${SLURM_JOB_ID:-local}" +echo "[tilert-run_node] ROLE=$TILERT_ROLE host=$(hostname) DECODE_HOST=$DECODE_HOST PREFILL_HOST=$PREFILL_HOST" + +log_and_run_bg() { + local label="$1" logfile="$2"; shift 2 + local _xtrace=0; [[ $- == *x* ]] && _xtrace=1 + { set +x; } 2>/dev/null + { printf '===== [%s] %s =====\n' "$label" "$(date '+%F %T')" + printf '[cmd]'; printf ' %q' "$@"; printf '\n' + printf '[cwd] %s\n[host] %s\n\n' "$PWD" "$(hostname)" + } | tee -a "$logfile" + "$@" >>"$logfile" 2>&1 & + LAST_BG_PID=$! + echo "[$label] pid=$LAST_BG_PID log=$logfile" + (( _xtrace )) && set -x + return 0 +} + +bench_result_stem() { + local conc="$1" + local pg=$(( ${PREFILL_TP:-8} * ${PREFILL_NUM_WORKERS:-1} )) + local dg=$(( ${DECODE_TP:-8} * ${DECODE_NUM_WORKERS:-1} )) + printf '%s_c%s_gpus_%s_ctx_%s_gen_%s' \ + "${RESULT_FILENAME}" "$conc" "$(( pg + dg ))" "$pg" "$dg" +} + +rdma_preflight() { + local warn=0 + echo "[rdma] role=$TILERT_ROLE UCX_NET_DEVICES=${UCX_NET_DEVICES:-} UCX_MEMTYPE_CACHE=${UCX_MEMTYPE_CACHE:-} UCX_MEMTYPE_REG_WHOLE=${UCX_MEMTYPE_REG_WHOLE:-}" + + local uverbs=(/dev/infiniband/uverbs*) + if [[ -e "${uverbs[0]}" ]]; then + echo "[rdma] verbs devices: ${uverbs[*]}" + else + echo "[rdma] WARNING: /dev/infiniband/uverbs* missing -- the container has no RDMA device nodes." >&2 + echo "[rdma] docker: add --device /dev/infiniband; pyxis: needs cluster-side passthrough" >&2 + warn=1 + fi + + local ml; ml="$(ulimit -l 2>/dev/null)" + if [[ "$ml" == "unlimited" ]]; then + echo "[rdma] memlock: unlimited" + else + echo "[rdma] WARNING: memlock=$ml (not unlimited) -- pinning memory for RDMA may fail." >&2 + echo "[rdma] docker: add --cap-add CAP_IPC_LOCK (or --ulimit memlock=-1)" >&2 + warn=1 + fi + + if command -v ibv_devices >/dev/null 2>&1; then + echo "[rdma] ibv_devices:"; ibv_devices 2>&1 | sed 's/^/[rdma] /' + fi + + if (( warn )) && [[ "${TILERT_RDMA_STRICT:-0}" == "1" ]]; then + echo "[rdma] TILERT_RDMA_STRICT=1 and preflight did not fully pass -- aborting" >&2 + return 1 + fi + return 0 +} + +stage_tokenizer_files() { + local staged=0 f b + for f in "$MODEL_PATH"/*; do + [[ -f "$f" ]] || continue + b="$(basename "$f")" + [[ "$b" == *.safetensors ]] && continue + [[ "$b" == "model.safetensors.index.json" ]] && continue + [[ -e "$TILERT_WEIGHTS_DIR/$b" ]] && continue + cp -p "$f" "$TILERT_WEIGHTS_DIR/$b" && staged=$((staged+1)) + done + echo "[stage_tokenizer] staged $staged auxiliary file(s) from $MODEL_PATH" + local missing=() + [[ -f "$TILERT_WEIGHTS_DIR/chat_template.jinja" ]] || missing+=(chat_template.jinja) + [[ -f "$TILERT_WEIGHTS_DIR/tokenizer_config.json" || -f "$TILERT_WEIGHTS_DIR/tokenizer.json" ]] \ + || missing+=("tokenizer.json/tokenizer_config.json") + if (( ${#missing[@]} )); then + echo "[stage_tokenizer] ERROR: $TILERT_WEIGHTS_DIR is missing ${missing[*]}" + echo "[stage_tokenizer] decode_server loads the tokenizer and chat template from that directory." + echo "[stage_tokenizer] Check that MODEL_PATH=$MODEL_PATH is an HF directory containing the tokenizer." + return 1 + fi + return 0 +} + +convert_weights() { + local index_json="$TILERT_WEIGHTS_DIR/model.safetensors.index.json" + if [[ -f "$index_json" ]]; then + echo "[weight_converter] cache hit (index.json present), skipping conversion: $TILERT_WEIGHTS_DIR" + return 0 + fi + + mkdir -p "$TILERT_WEIGHTS_DIR" + exec 9>"$TILERT_WEIGHTS_DIR/.convert.lock" + flock -w "${TILERT_CONVERT_LOCK_WAIT:-21600}" 9 || { + echo "[weight_converter] timed out waiting for the conversion lock (another job still converting?)"; return 1; } + if [[ -f "$index_json" ]]; then + echo "[weight_converter] cache produced by a concurrent job, skipping conversion"; exec 9>&-; return 0 + fi + if [[ -n "$(ls -A "$TILERT_WEIGHTS_DIR" 2>/dev/null | grep -v '^\.convert\.lock$')" ]]; then + echo "[weight_converter] found leftovers without index.json (previous conversion incomplete), cleaning and re-converting" + find "$TILERT_WEIGHTS_DIR" -mindepth 1 ! -name '.convert.lock' -delete + fi + + echo "[weight_converter] $MODEL_PATH -> $TILERT_WEIGHTS_DIR (model_type=$TILERT_MODEL_TYPE)" + "${PY:-python}" -m tilert.models.preprocess.weight_converter \ + --model_type "$TILERT_MODEL_TYPE" --model_dir "$MODEL_PATH" --save_dir "$TILERT_WEIGHTS_DIR" + local rc=$? + exec 9>&- + if [[ $rc -ne 0 || ! -f "$index_json" ]]; then + echo "[weight_converter] conversion failed (rc=$rc, no index.json produced): $TILERT_WEIGHTS_DIR" + return 1 + fi + echo "[weight_converter] conversion done and cached: $TILERT_WEIGHTS_DIR" +} + +start_decode() { + local cmd=("${PY:-python}" -m tilert.pd_vllm.decode_server + --engine tilert --model "$MODEL_NAME" + --model-weights-dir "$TILERT_WEIGHTS_DIR" + --max-seq-len "$MAX_MODEL_LEN" + --kv-cache-dtype "$DECODE_KV_DTYPE" --transport "$KV_P2P_TRANSFER" + --ctrl-port "$DECODE_CTRL_PORT" --http-port "$DECODE_HTTP_PORT" + "${DECODE_MTP[@]}") + log_and_run_bg decode "$BENCHMARK_LOGS_DIR/tilert_decode.log" "${cmd[@]}" + DECODE_PID=$LAST_BG_PID +} + +start_prefill() { + local cmd=(vllm serve "$MODEL_PATH" + --served-model-name "$MODEL_NAME" --port "$PREFILL_PORT" + --tensor-parallel-size "$PREFILL_TP" --max-model-len "$MAX_MODEL_LEN" + --enforce-eager --trust-remote-code --return-tokens-as-token-ids + --gpu-memory-utilization "$GPU_MEM_UTIL" --kv-cache-dtype "$PREFILL_KV_DTYPE" + "${PREFILL_SPEC[@]}" + --kv-transfer-config "{\"kv_connector\":\"TileRTConnector\",\"kv_connector_module_path\":\"tilert.pd_vllm.prefill_connector\",\"kv_role\":\"kv_producer\",\"kv_connector_extra_config\":{\"tilert_host\":\"$DECODE_HOST\",\"tilert_ctrl_port\":$DECODE_CTRL_PORT,\"tilert_model\":\"$MODEL_NAME\",\"tilert_max_seq_len\":$MAX_MODEL_LEN,\"tilert_transport\":\"$KV_P2P_TRANSFER\"}}") + log_and_run_bg prefill "$BENCHMARK_LOGS_DIR/tilert_prefill.log" "${cmd[@]}" + PREFILL_PID=$LAST_BG_PID +} + +start_router() { + local cmd=(env CUDA_VISIBLE_DEVICES= "${PY:-python}" -m tilert.pd_vllm.pd_router + --vllm-url "http://$PREFILL_HOST:$PREFILL_PORT" + --decode "$DECODE_HOST:$DECODE_CTRL_PORT:$DECODE_HTTP_PORT" + --port "$ROUTER_PORT" --model-path "$MODEL_PATH" --parser "$TILERT_PARSER") + log_and_run_bg router "$BENCHMARK_LOGS_DIR/tilert_router.log" "${cmd[@]}" + ROUTER_PID=$LAST_BG_PID +} + +wait_for_tcp() { + local host="$1" port="$2" deadline=$(( SECONDS + ${3:-600} )) + local _xtrace=0; [[ $- == *x* ]] && _xtrace=1 + { set +x; } 2>/dev/null + local rc=0 + until (exec 3<>"/dev/tcp/$host/$port") 2>/dev/null; do + if [[ $SECONDS -ge $deadline ]]; then + echo "[wait_for_tcp] timeout $host:$port after ${3:-600}s"; rc=1; break + fi + sleep 5 + done + [[ $rc -eq 0 ]] && { exec 3>&- 2>/dev/null || true; echo "[wait_for_tcp] $host:$port ready"; } + (( _xtrace )) && set -x + return $rc +} + +run_bench_and_eval() { + wait_for_server_ready --port "$ROUTER_PORT" \ + --server-log "$BENCHMARK_LOGS_DIR/tilert_router.log" --server-pid "$ROUTER_PID" + local rc=0 conc np + for conc in $CONC_LIST; do + np=$(( conc * 10 )) + [[ "$np" -lt 16 ]] && np=16 + run_benchmark_serving \ + --bench-serving-dir /workspace \ + --model "$MODEL_NAME" --port "$ROUTER_PORT" \ + --backend openai-chat --endpoint /v1/chat/completions \ + --input-len "$ISL" --output-len "$OSL" \ + --random-range-ratio "$RANDOM_RANGE_RATIO" \ + --num-prompts "$np" --max-concurrency "$conc" \ + --use-chat-template --server-pid "$ROUTER_PID" \ + --tokenizer "$MODEL_PATH" --trust-remote-code \ + --result-filename "$(bench_result_stem "$conc")" --result-dir "$RESULT_DIR" \ + || { rc=$?; echo "[bench] WARNING: conc=$conc failed/timed out (rc=$rc)"; } + done + if [[ "${RUN_EVAL}" = "true" ]]; then + if [[ -n "${EVAL_CONC:-}" ]]; then + export EVAL_CONCURRENT_REQUESTS="$EVAL_CONC" + else + export EVAL_CONCURRENT_REQUESTS="$(tr ' ' '\n' <<< "$CONC_LIST" | sort -n | tail -1)" + fi + export CONC="$EVAL_CONCURRENT_REQUESTS" + run_eval --framework lm-eval --port "$ROUTER_PORT" + append_lm_eval_summary + fi + return $rc +} + +# shellcheck source=./setup_deps.sh +source "$(dirname "$0")/setup_deps.sh" + +set -x +case "$TILERT_ROLE" in + decode) + rdma_preflight || exit 1 + convert_weights || exit 1 + stage_tokenizer_files || exit 1 + start_decode + { set +x; } 2>/dev/null + while kill -0 "$DECODE_PID" 2>/dev/null; do + [[ -f "$DONE_SENTINEL" ]] && break + sleep 5 + done + if [[ -f "$DONE_SENTINEL" ]]; then + echo "[decode] done sentinel received, shutting down"; kill "$DECODE_PID" 2>/dev/null || true; exit 0 + fi + echo "[decode] decode_server exited early (see $BENCHMARK_LOGS_DIR/tilert_decode.log)"; exit 1 + ;; + prefill) + rdma_preflight || exit 1 + rm -f "$DONE_SENTINEL" + wait_for_tcp "$DECODE_HOST" "$DECODE_CTRL_PORT" "$DECODE_WAIT" \ + || echo "[prefill] WARNING: timed out waiting for the decode ctrl port ($DECODE_HOST:$DECODE_CTRL_PORT), starting anyway" + start_prefill + wait_for_tcp "$PREFILL_HOST" "$PREFILL_PORT" "${PREFILL_WAIT:-3600}" \ + || echo "[prefill] WARNING: timed out waiting for the vLLM port ($PREFILL_HOST:$PREFILL_PORT), continuing (see $BENCHMARK_LOGS_DIR/tilert_prefill.log)" + start_router + run_bench_and_eval; BENCH_RC=$? + touch "$DONE_SENTINEL" + kill "$ROUTER_PID" "$PREFILL_PID" 2>/dev/null || true + exit $BENCH_RC + ;; + *) + echo "unknown ROLE=$TILERT_ROLE"; exit 2 ;; +esac diff --git a/benchmarks/multi_node/tilert_utils/setup_deps.sh b/benchmarks/multi_node/tilert_utils/setup_deps.sh new file mode 100644 index 0000000000..047087af2d --- /dev/null +++ b/benchmarks/multi_node/tilert_utils/setup_deps.sh @@ -0,0 +1,142 @@ +#!/bin/bash + +TILERT_VERSION="${TILERT_VERSION:-0.1.5.post2}" +TILERT_PIP_INDEX_URL="${TILERT_PIP_INDEX_URL:-}" + +TILERT_HTTP_DEPS="${TILERT_HTTP_DEPS:-fastapi uvicorn httpx}" +TILERT_NIXL_VERSION="${TILERT_NIXL_VERSION:-1.3.1}" +TILERT_TRANSPORT_DEPS="${TILERT_TRANSPORT_DEPS:-nixl==$TILERT_NIXL_VERSION}" + +_SETUP_INSTALLED=() + +activate_tilert_env() { + local env_dir=/opt/conda/envs/tilert + [[ -d "$env_dir" ]] || return 0 + if [[ "$(command -v python)" == "$env_dir/bin/python" ]]; then + echo "[SETUP] conda env 'tilert' already active" + return 0 + fi + echo "[SETUP] activating conda env 'tilert' (enroot does not run the image ENTRYPOINT)" + # shellcheck disable=SC1091 + if [[ -f /opt/conda/etc/profile.d/conda.sh ]]; then + . /opt/conda/etc/profile.d/conda.sh && conda activate tilert + fi + [[ "$(command -v python)" == "$env_dir/bin/python" ]] || export PATH="$env_dir/bin:$PATH" + echo "[SETUP] python -> $(command -v python)" +} + +_installed_version() { + "$PY" - "$1" <<'PY' 2>/dev/null +import sys +from importlib.metadata import version, PackageNotFoundError +try: + print(version(sys.argv[1])) +except PackageNotFoundError: + pass +PY +} + +_pip_args() { + local a=(--quiet --no-cache-dir) + [[ -n "$TILERT_PIP_INDEX_URL" ]] && a+=(--index-url "$TILERT_PIP_INDEX_URL") + printf '%s\n' "${a[@]}" +} + +install_tilert_decode() { + mapfile -t _pa < <(_pip_args) + local have; have="$(_installed_version tilert)" + if [[ "$have" == "$TILERT_VERSION" ]]; then + echo "[SETUP] tilert $have already installed, skipping" + else + [[ -n "$have" ]] && echo "[SETUP] tilert $have installed, switching to pinned $TILERT_VERSION" + echo "[SETUP] installing tilert==$TILERT_VERSION (official PyPI release wheel)" + "$PY" -m pip install "${_pa[@]}" "tilert==$TILERT_VERSION" || { + echo "[SETUP] ERROR: failed to install tilert==$TILERT_VERSION"; exit 1; } + have="$(_installed_version tilert)" + [[ "$have" == "$TILERT_VERSION" ]] || { + echo "[SETUP] ERROR: still not $TILERT_VERSION after install (actual: ${have:-not installed})"; exit 1; } + _SETUP_INSTALLED+=("tilert==$TILERT_VERSION") + fi + _install_missing "uvicorn" "$TILERT_HTTP_DEPS" + _install_missing "nixl" "$TILERT_TRANSPORT_DEPS" + local tv; tv="$(_installed_version transformers)" + if [[ -n "$tv" && "${tv%%.*}" -lt 5 ]]; then + echo "[SETUP] transformers $tv < 5 -- cannot load the official checkpoint's TokenizersBackend, upgrading" + "$PY" -m pip install "${_pa[@]}" -U "transformers>=5.4.0" || { + echo "[SETUP] ERROR: failed to upgrade transformers"; exit 1; } + _SETUP_INSTALLED+=("transformers>=5.4.0(upgrade from $tv)") + fi + "$PY" -c "import tilert.pd_vllm.decode_server" 2>/dev/null || { + echo "[SETUP] ERROR: import tilert.pd_vllm.decode_server failed; actual error:" + "$PY" -c "import tilert.pd_vllm.decode_server" 2>&1 | tail -3 + exit 1; } + echo "[SETUP] tilert.pd_vllm.decode_server imports OK" +} + +_install_missing() { + local probe="$1" pkgs="$2" + [[ -n "$pkgs" ]] || return 0 + if "$PY" -c "import $probe" 2>/dev/null; then + echo "[SETUP] $probe already present, skipping ($pkgs)" + return 0 + fi + echo "[SETUP] installing $pkgs (probe module $probe missing)" + mapfile -t _pa < <(_pip_args) + # shellcheck disable=SC2086 + "$PY" -m pip install "${_pa[@]}" $pkgs || { + echo "[SETUP] ERROR: failed to install: $pkgs"; exit 1; } + _SETUP_INSTALLED+=("$pkgs") +} + +install_tilert_prefill() { + local vllm_v; vllm_v="$(_installed_version vllm)" + if [[ -z "$vllm_v" ]]; then + echo "[SETUP] ERROR: no vLLM in the prefill image." + echo "[SETUP] prefill needs an image with V1 disaggregation + GLM-5/5.1 (DSA) +" + echo "[SETUP] --kv-cache-dtype fp8_ds_mla support; the official tilert image has no vLLM," + echo "[SETUP] and its dependency set conflicts with vLLM, so P and D must use different images." + exit 1 + fi + echo "[SETUP] prefill-side vLLM $vllm_v" + local have; have="$(_installed_version tilert)" + if [[ "$have" == "$TILERT_VERSION" ]]; then + echo "[SETUP] tilert $have already installed, skipping" + else + echo "[SETUP] installing tilert==$TILERT_VERSION --no-deps (connector plugin only; leaves transformers untouched)" + mapfile -t _pa < <(_pip_args) + "$PY" -m pip install "${_pa[@]}" --no-deps "tilert==$TILERT_VERSION" || { + echo "[SETUP] ERROR: failed to install tilert==$TILERT_VERSION (--no-deps)"; exit 1; } + have="$(_installed_version tilert)" + [[ "$have" == "$TILERT_VERSION" ]] || { + echo "[SETUP] ERROR: still not $TILERT_VERSION after install (actual: ${have:-not installed})"; exit 1; } + _SETUP_INSTALLED+=("tilert==$TILERT_VERSION(--no-deps)") + fi + _install_missing "nixl" "$TILERT_TRANSPORT_DEPS" + "$PY" -c "import tilert.pd_vllm.prefill_connector" 2>/dev/null || { + echo "[SETUP] WARN: import tilert.pd_vllm.prefill_connector failed (vLLM will report again when loading the connector plugin):" + "$PY" -c "import tilert.pd_vllm.prefill_connector" 2>&1 | tail -3; } +} + +resolve_python() { + if [[ -n "${PY:-}" ]] && command -v "$PY" >/dev/null 2>&1; then : + else + PY="" + for c in python python3; do command -v "$c" >/dev/null 2>&1 && { PY="$c"; break; }; done + fi + [[ -n "$PY" ]] || { echo "[SETUP] ERROR: neither python nor python3 found"; exit 1; } + export PY + echo "[SETUP] interpreter PY=$PY ($(command -v "$PY"))" +} + +activate_tilert_env +resolve_python +case "${TILERT_ROLE:-}" in + decode) install_tilert_decode ;; + prefill) install_tilert_prefill ;; + *) echo "[SETUP] ERROR: unknown ROLE='${TILERT_ROLE:-}'"; exit 1 ;; +esac +if (( ${#_SETUP_INSTALLED[@]} )); then + echo "[SETUP] installed this run: ${_SETUP_INSTALLED[*]}" +else + echo "[SETUP] nothing to install (dependencies already satisfied)" +fi diff --git a/benchmarks/multi_node/tilert_utils/submit.sh b/benchmarks/multi_node/tilert_utils/submit.sh new file mode 100755 index 0000000000..b63e97bef7 --- /dev/null +++ b/benchmarks/multi_node/tilert_utils/submit.sh @@ -0,0 +1,74 @@ +#!/usr/bin/env bash +set -x +NODES=$(( ${PREFILL_NODES:-1} + ${DECODE_NODES:-1} )) +GPUS_PER_NODE="${GPUS_PER_NODE:-${GPU_COUNT:-$(( ${PREFILL_TP:-8} > ${DECODE_TP:-8} ? ${PREFILL_TP:-8} : ${DECODE_TP:-8} ))}}" + +SQUASH_DIR="${B200_SQUASH_DIR:-/home/sa-shared/containers}" +{ mkdir -p "$SQUASH_DIR" 2>/dev/null && [[ -w "$SQUASH_DIR" ]]; } || SQUASH_DIR="$GITHUB_WORKSPACE/.container-squash" +mkdir -p "$SQUASH_DIR" +chmod a+rx "$SQUASH_DIR" || true + +DECODE_IMAGE="${DECODE_IMAGE:-$IMAGE}" +: "${PREFILL_IMAGE:?PREFILL_IMAGE is unset -- it must come from prefill.additional-settings in the master config, e.g. PREFILL_IMAGE=vllm/vllm-openai:v0.26.0}" + +squash_path() { echo "$SQUASH_DIR/$(echo "$1" | sed 's/[\/:@#]/_/g').sqsh"; } +DECODE_SQUASH="$(squash_path "$DECODE_IMAGE")" +PREFILL_SQUASH="$(squash_path "$PREFILL_IMAGE")" + +salloc --partition="$SLURM_PARTITION" --account="$SLURM_ACCOUNT" \ + --nodes="$NODES" --gres=gpu:"$GPUS_PER_NODE" --exclusive --mem=0 \ + --time="${SALLOC_TIME_LIMIT:-480}" --no-shell --job-name="$RUNNER_NAME" +JOB_ID=$(squeue --name="$RUNNER_NAME" -u "$USER" -h -o %A | head -n1) + +mapfile -t HOSTS < <(scontrol show hostnames "$(squeue -j "$JOB_ID" -h -o %N)") +[[ "${#HOSTS[@]}" -ge 2 ]] || { echo "expected >=2 nodes, got: ${HOSTS[*]}"; exit 1; } +export DECODE_HOST="${HOSTS[0]}" PREFILL_HOST="${HOSTS[1]}" + +import_image() { + local image_ref="$1" squash_file="$2" host="$3" + local image_key; image_key=$(echo "$image_ref" | sed 's/[\/:@#]/_/g') + local lock_file="$SQUASH_DIR/.locks/${image_key}.lock" + mkdir -p "$SQUASH_DIR/.locks" + srun --jobid="$JOB_ID" --nodelist="$host" --ntasks=1 bash -c " + export ENROOT_CACHE_PATH=\$HOME/.cache/enroot; mkdir -p \$ENROOT_CACHE_PATH + exec 9>\"$lock_file\"; flock -w 600 9 || exit 1 + unsquashfs -l \"$squash_file\" >/dev/null 2>&1 || enroot import -o \"$squash_file\" docker://$image_ref + " +} +import_image "$DECODE_IMAGE" "$DECODE_SQUASH" "$DECODE_HOST" || exit 1 +import_image "$PREFILL_IMAGE" "$PREFILL_SQUASH" "$PREFILL_HOST" || exit 1 + +export TILERT_WEIGHTS_DIR="${TILERT_WEIGHTS_DIR:-$HOME/.cache/tilert/${MODEL_PREFIX:-model}-${PRECISION:-fp8}-tilert-8shard}" +mkdir -p "$TILERT_WEIGHTS_DIR" + +run_role() { + local role="$1" host="$2" squash_file="$3" + # The tilert image bakes no NVIDIA_VISIBLE_DEVICES (unlike vllm-openai), and + # enroot's nvidia hook only injects the driver when it is set — without it the + # decode container has no libcuda and torch dies with "Found no NVIDIA driver". + # docker --gpus sets this implicitly, which is why the image works elsewhere. + # Exported here (not in --export) because the capabilities value contains a + # comma, which srun's --export parsing would split on. + export NVIDIA_VISIBLE_DEVICES=all NVIDIA_DRIVER_CAPABILITIES=compute,utility + srun --jobid="$JOB_ID" --nodelist="$host" --ntasks=1 \ + --container-image="$squash_file" \ + --container-mounts="$GITHUB_WORKSPACE:/workspace,$MODEL_PATH:$MODEL_PATH,$TILERT_WEIGHTS_DIR:$TILERT_WEIGHTS_DIR" \ + --container-workdir=/workspace --no-container-entrypoint \ + --export=ALL,TILERT_ROLE="$role",DECODE_HOST="$DECODE_HOST",PREFILL_HOST="$PREFILL_HOST",PORT="${PORT:-8888}" \ + bash "/workspace/benchmarks/multi_node/tilert_utils/run_node.sh" +} + +run_role decode "$DECODE_HOST" "$DECODE_SQUASH" & +DECODE_SRUN_PID=$! + +run_role prefill "$PREFILL_HOST" "$PREFILL_SQUASH" +PREFILL_RC=$? + +for _ in $(seq 1 "${TILERT_DECODE_DRAIN:-60}"); do + kill -0 "$DECODE_SRUN_PID" 2>/dev/null || break + sleep 1 +done +kill -0 "$DECODE_SRUN_PID" 2>/dev/null && { echo "[submit] decode srun did not exit on its own, killing it"; kill "$DECODE_SRUN_PID" 2>/dev/null; } +wait "$DECODE_SRUN_PID" 2>/dev/null || true + +exit "$PREFILL_RC" diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index 6e0cf3e272..775c8953ce 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -7931,3 +7931,58 @@ qwen3.5-fp8-gb300-dynamo-sglang-mtp: tp: 16 ep: 16 dp-attn: true + +# tileRT (github.com/tile-ai/TileRT) -- vLLM prefill + TileRT decode PD disaggregation; single point at bs=1. +glm5.1-fp8-b200-tilert: + image: ghcr.io/tile-ai/tilert:0.1.5 + model: zai-org/GLM-5.1-FP8 + model-prefix: glm5.1 + runner: b200-multinode + precision: fp8 + framework: tilert + router: { name: tilert-pd-router, version: "0.1.5.post2" } + multinode: true + disagg: true + kv-p2p-transfer: nixl + scenarios: + fixed-seq-len: + - isl: 1024 + osl: 1024 + search-space: + - spec-decoding: "mtp" + conc-list: [1] + prefill: + num-worker: 1 + tp: 8 + ep: 1 + dp-attn: false + additional-settings: + - "PREFILL_IMAGE=vllm/vllm-openai:v0.26.0" + - "PREFILL_NODES=1" + decode: + num-worker: 1 + tp: 8 + ep: 1 + dp-attn: false + additional-settings: + - "DECODE_NODES=1" + - isl: 8192 + osl: 1024 + search-space: + - spec-decoding: "mtp" + conc-list: [1] + prefill: + num-worker: 1 + tp: 8 + ep: 1 + dp-attn: false + additional-settings: + - "PREFILL_IMAGE=vllm/vllm-openai:v0.26.0" + - "PREFILL_NODES=1" + decode: + num-worker: 1 + tp: 8 + ep: 1 + dp-attn: false + additional-settings: + - "DECODE_NODES=1" diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 2abd59df46..c9decc201c 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -5556,6 +5556,15 @@ - "Allow high-concurrency full-context AgentX jobs to complete the canonical warmup and one-hour profile with a 12-hour Slurm limit and 13-hour workflow envelope." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2404 +- config-keys: + - glm5.1-fp8-b200-tilert + description: + - "Add GLM-5.1 FP8 B200 prefill/decode-disaggregated benchmark via TileRT: vLLM prefill (TileRTConnector, kv_producer) + TileRT decode_server + OpenAI-compatible pd_router, framework=tilert (non-dynamo custom runtime, decode-only engine)" + - "Image ghcr.io/tile-ai/tilert:0.1.5 (tilert 0.1.5.post2 installed at container start); commands aligned to TileRT README Topology A -- NIXL KV transfer, --kv-cache-dtype fp8_ds_mla (prefill) <-> fp8 (decode), max-seq-len 202752; MTP speculative-config wired via spec-decoding=mtp" + - "Topology: 1 prefill node (TP8) + 1 decode node (TP8), each 8xB200 exclusive; TileRT decode is bs=1 only so conc-list is a single point [1], ISL 1k/8k OSL 1k" + - "Runner: launch_b200-dgxc.sh tilert early-return branch (zero impact on the dynamo path); tilert_utils/submit.sh issues two srun --ntasks=1, one per role, because prefill and decode need different container images; roles are dispatched by the TILERT_ROLE it exports, and torn down across nodes via a sentinel file on the shared /workspace" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXX + - config-keys: - qwen3.5-fp8-b200-sglang-agentic-mtp scenario-type: diff --git a/runners/launch_b200-dgxc.sh b/runners/launch_b200-dgxc.sh index edec0f1289..778eac1e51 100644 --- a/runners/launch_b200-dgxc.sh +++ b/runners/launch_b200-dgxc.sh @@ -43,6 +43,14 @@ elif [[ $MODEL_PREFIX == "qwen3.5" && $PRECISION == "fp4" ]]; then elif [[ $MODEL_PREFIX == "glm5" && $PRECISION == "fp8" ]]; then export MODEL_PATH="/lustre/fsw/models/GLM-5-FP8" export SRT_SLURM_MODEL_PREFIX="glm5-fp8" +elif [[ $MODEL_PREFIX == "glm5.1" && $PRECISION == "fp8" ]]; then + # GLM-5.1 retired in July and its weights were cleaned out of the + # SRE-owned (root-only) /lustre/fsw/models tree, so this checkpoint is + # staged on the sa-shared-writable home Lustre mount instead. That mount + # is compute-visible at the same path, and the launcher bind-mounts + # $MODEL_PATH by name into the container. + export MODEL_PATH="/home/sa-shared/models/GLM-5.1-FP8" + export SRT_SLURM_MODEL_PREFIX="glm5.1-fp8" elif [[ $MODEL_PREFIX == "glm5" && $PRECISION == "fp4" ]]; then export MODEL_PATH="/lustre/fsw/models/GLM-5-NVFP4" export SRT_SLURM_MODEL_PREFIX="glm5-fp4" @@ -111,6 +119,21 @@ fi export AIPERF_MMAP_CACHE_HOST_PATH="/lustre/fsw/gharunners/aiperf-cache" if [[ "$IS_MULTINODE" == "true" ]]; then + if [[ "$FRAMEWORK" == "tilert" ]]; then + export SLURM_PARTITION SLURM_ACCOUNT + export TILERT_WEIGHTS_DIR="${TILERT_WEIGHTS_DIR:-/lustre/fsw/gharunners/models/${MODEL_PREFIX}-${PRECISION}-tilert-8shard}" + # These nodes expose eight RoCE HCAs, mlx5_0..mlx5_7 (all PORT_ACTIVE, + # link_layer Ethernet); there is no mlx5_10/mlx5_11 here. Verified + # with ibv_devinfo inside a pyxis container on an allocated gpu-2 node. + export UCX_NET_DEVICES="${UCX_NET_DEVICES:-mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1,mlx5_4:1,mlx5_5:1,mlx5_6:1,mlx5_7:1}" + export UCX_MEMTYPE_CACHE="${UCX_MEMTYPE_CACHE:-n}" + export UCX_MEMTYPE_REG_WHOLE="${UCX_MEMTYPE_REG_WHOLE:-n}" + TILERT_DISAGG="$GITHUB_WORKSPACE/benchmarks/multi_node/${EXP_NAME%%_*}_${PRECISION}_b200_${FRAMEWORK}-disagg.sh" + [[ -f "$TILERT_DISAGG" ]] || { echo "tilert disagg script not found: $TILERT_DISAGG"; exit 1; } + exec bash "$TILERT_DISAGG" + exit 1 + fi + # Validate framework if [[ $FRAMEWORK != "dynamo-sglang" && $FRAMEWORK != "dynamo-trt" && $FRAMEWORK != "dynamo-vllm" ]]; then echo "Unsupported framework: $FRAMEWORK. Supported frameworks are: dynamo-trt, dynamo-sglang, dynamo-vllm"