Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 5 additions & 3 deletions .github/configs/nvidia-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -2044,7 +2044,7 @@ qwen3.5-fp8-b200-sglang:
- { tp: 4, ep: 1, conc-start: 4, conc-end: 256 }

qwen3.5-fp4-b200-sglang:
image: lmsysorg/sglang:nightly-dev-20260402-d7256eb6
image: lmsysorg/sglang:nightly-dev-20260422-de962f32
model: nvidia/Qwen3.5-397B-A17B-NVFP4
Comment thread
Ankur-singh marked this conversation as resolved.
model-prefix: qwen3.5
runner: b200
Expand All @@ -2056,11 +2056,13 @@ qwen3.5-fp4-b200-sglang:
- isl: 1024
osl: 1024
search-space:
- { tp: 4, ep: 1, conc-start: 4, conc-end: 128 }
- { tp: 4, ep: 1, conc-start: 4, conc-end: 4 }
- { tp: 2, ep: 1, conc-start: 4, conc-end: 128 }
- isl: 8192
osl: 1024
search-space:
- { tp: 4, ep: 1, conc-start: 4, conc-end: 128 }
- { tp: 4, ep: 1, conc-start: 4, conc-end: 4 }
- { tp: 2, ep: 1, conc-start: 4, conc-end: 128 }

qwen3.5-fp4-b200-sglang-mtp:
image: lmsysorg/sglang:nightly-dev-20260402-d7256eb6
Expand Down
48 changes: 15 additions & 33 deletions benchmarks/single_node/qwen3.5_fp4_b200.sh
Original file line number Diff line number Diff line change
Expand Up @@ -20,56 +20,38 @@ nvidia-smi

hf download "$MODEL"

export NCCL_NVLS_ENABLE=1
export SGL_ENABLE_JIT_DEEPGEMM=false
export SGLANG_ENABLE_FLASHINFER_GEMM=true
export PYTHONUNBUFFERED=1

SERVER_LOG=/workspace/server.log
PORT=${PORT:-8888}

# Default: recv every ~10 requests; if CONC >= 16, relax to ~30 requests between scheduler recv polls.
if [[ $CONC -ge 16 ]]; then
SCHEDULER_RECV_INTERVAL=30
else
SCHEDULER_RECV_INTERVAL=10
fi

MEM_FRAC_STATIC=0.85
CHUNKED_PREFILL_SIZE=32768
MAX_PREFILL_TOKENS=32768
CUDA_GRAPH_MAX_BATCH_SIZE=$CONC
MAX_RUNNING_REQUESTS=128
CONTEXT_LENGTH=$((ISL + OSL + 20))
if [ "${EVAL_ONLY}" = "true" ]; then
setup_eval_context
CONTEXT_LENGTH="$EVAL_MAX_MODEL_LEN"
fi

if [[ $TP -eq 8 ]]; then
EXTRA_ARGS="--enable-flashinfer-allreduce-fusion"
else
EXTRA_ARGS=""
fi

echo "SCHEDULER_RECV_INTERVAL: $SCHEDULER_RECV_INTERVAL, CONC: $CONC, ISL: $ISL, OSL: $OSL"

# Start GPU monitoring (power, temperature, clocks every second)
start_gpu_monitor

set -x
PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path=$MODEL --host=0.0.0.0 --port=$PORT \
--trust-remote-code \
--tensor-parallel-size=$TP --data-parallel-size=1 --ep-size $EP_SIZE \
--quantization modelopt_fp4 --fp4-gemm-backend flashinfer_cutlass \
--tensor-parallel-size=$TP --data-parallel-size=1 --expert-parallel-size=$EP_SIZE \
--enable-symm-mem \
--disable-radix-cache \
--quantization modelopt_fp4 \
--kv-cache-dtype fp8_e4m3 \
--mamba-ssm-dtype bfloat16 \
--cuda-graph-max-bs $CUDA_GRAPH_MAX_BATCH_SIZE --max-running-requests $MAX_RUNNING_REQUESTS \
--mem-fraction-static $MEM_FRAC_STATIC --chunked-prefill-size $CHUNKED_PREFILL_SIZE --max-prefill-tokens $MAX_PREFILL_TOKENS \
--context-length $CONTEXT_LENGTH --disable-radix-cache \
--attention-backend trtllm_mha --moe-runner-backend flashinfer_trtllm \
$EXTRA_ARGS --scheduler-recv-interval $SCHEDULER_RECV_INTERVAL \
--tokenizer-worker-num 6 --stream-interval 30 > $SERVER_LOG 2>&1 &
--attention-backend trtllm_mha \
--moe-runner-backend flashinfer_trtllm \
--cuda-graph-max-bs $CONC \
--max-prefill-tokens 16384 \
--chunked-prefill-size 16384 \
--mem-fraction-static 0.8 \
--stream-interval 50 \
--scheduler-recv-interval $( [[ $CONC -gt 4 ]] && echo 30 || echo 10 ) \
--tokenizer-worker-num 6 \
--tokenizer-path $MODEL \
--context-length $CONTEXT_LENGTH > $SERVER_LOG 2>&1 &

Comment thread
Ankur-singh marked this conversation as resolved.
SERVER_PID=$!

Expand Down
9 changes: 9 additions & 0 deletions perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -2078,3 +2078,12 @@
- "Removed redundant transformer installation"
- "Optimization model serves configs"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/1252

- config-keys:
- qwen3.5-fp4-b200-sglang
description:
- "Update image to lmsysorg/sglang:nightly-dev-20260422-de962f32"
- "Fix tp:2 search-space: ep 2 -> 1"
- "Dynamic scheduler-recv-interval: 30 for CONC>4, 10 otherwise"
- "Remove --max-running-requests, reduce prefill/chunked from 81920 to 16384"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/1018
Loading