Skip to content
Merged
Show file tree
Hide file tree
Changes from 1 commit
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
@@ -0,0 +1,116 @@
name: "minimax-m3-vllm-disagg-gb300-1p1d-dep2-dep8-mxfp8-8k1k-eagle3-c64"

model:
path: "minimax-m3-mxfp8"
container: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7"
precision: "fp8"

identity:
model:
repo: "MiniMaxAI/MiniMax-M3-MXFP8"
revision: "c5454eb03678d8710e54a4e0fc681b9f3b4a3dba"
container:
image: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7"
frameworks:
dynamo: "1.4.0.dev20260730"
vllm: "0.26.1rc1.dev255+g5e35a6f4f"

dynamo:
install: true
version: "1.4.0.dev20260730"
request_plane: "nats"

health_check:
max_attempts: 720
interval_seconds: 10

sbatch_directives:
mem: "0"
cpus-per-task: "72"

srun_options:
mem: "0"

resources:
gpu_type: "gb300"
gpus_per_node: 4
prefill_nodes: 1
decode_nodes: 2
prefill_workers: 1
decode_workers: 1
gpus_per_prefill: 2
gpus_per_decode: 8

frontend:
type: "dynamo"
enable_multiple_frontends: false

backend:
type: "vllm"
connector: null

prefill_environment: &worker-environment
VLLM_ENGINE_READY_TIMEOUT_S: "3600"
VLLM_FLOAT32_MATMUL_PRECISION: "high"
VLLM_FLASHINFER_ALLREDUCE_BACKEND: "mnnvl"
UCX_CUDA_IPC_ENABLE_MNNVL: "y"
UCX_MODULE_DIR: "/usr/local/lib/python3.12/dist-packages/nixl_cu13.libs/ucx"
UCX_RNDV_PIPELINE_ERROR_HANDLING: "y"
NCCL_CUMEM_ENABLE: "1"
NCCL_MNNVL_ENABLE: "1"
NCCL_NVLS_ENABLE: "1"

decode_environment: *worker-environment

vllm_config:
prefill:
no-enable-flashinfer-autotune: true
kv-transfer-config: '{"kv_connector": "NixlConnector", "kv_role": "kv_both"}'
attention-config: '{"backend": "FLASHINFER", "use_trtllm_attention": true, "indexer_kv_dtype": "fp8"}'
speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}'
tensor-parallel-size: 1
data-parallel-size: 2
data-parallel-rpc-port: 13345
enable-expert-parallel: true
trust-remote-code: true
no-enable-prefix-caching: true
block-size: 128
gpu-memory-utilization: 0.90
max-model-len: 9472
language-model-only: true
kv-cache-dtype: "fp8"
stream-interval: 32
max-cudagraph-capture-size: 2048
max-num-batched-tokens: 16384

decode:
no-enable-flashinfer-autotune: true
kv-transfer-config: '{"kv_connector": "NixlConnector", "kv_role": "kv_both"}'
attention-config: '{"backend": "FLASHINFER", "use_trtllm_attention": true, "indexer_kv_dtype": "fp8", "minimax_m3_msa_decode_backend": "cutlass"}'
speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}'
tensor-parallel-size: 1
data-parallel-size: 8
data-parallel-rpc-port: 13345
enable-expert-parallel: true
trust-remote-code: true
no-enable-prefix-caching: true
block-size: 128
gpu-memory-utilization: 0.90
max-model-len: 9472
language-model-only: true
kv-cache-dtype: "fp8"
stream-interval: 32
max-num-seqs: 1024
max-num-batched-tokens: 16384
max-cudagraph-capture-size: 2048
cudagraph_mode: FULL_DECODE_ONLY

benchmark:
type: "sa-bench"
isl: 8192
osl: 1024
concurrencies: "64"
req_rate: "inf"
num_warmup_mult: 2
random_range_ratio: 0.8
use_chat_template: true
Original file line number Diff line number Diff line change
@@ -0,0 +1,114 @@
name: "minimax-m3-vllm-disagg-gb300-1p1d-dep2-tp4-mxfp8-8k1k-eagle3-c1"

model:
path: "minimax-m3-mxfp8"
container: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7"
precision: "fp8"

identity:
model:
repo: "MiniMaxAI/MiniMax-M3-MXFP8"
revision: "c5454eb03678d8710e54a4e0fc681b9f3b4a3dba"
container:
image: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7"
frameworks:
dynamo: "1.4.0.dev20260730"
vllm: "0.26.1rc1.dev255+g5e35a6f4f"

dynamo:
install: true
version: "1.4.0.dev20260730"
request_plane: "nats"

health_check:
max_attempts: 720
interval_seconds: 10

sbatch_directives:
mem: "0"
cpus-per-task: "72"

srun_options:
mem: "0"

resources:
gpu_type: "gb300"
gpus_per_node: 4
prefill_nodes: 1
decode_nodes: 1
prefill_workers: 1
decode_workers: 1
gpus_per_prefill: 2
gpus_per_decode: 4

frontend:
type: "dynamo"
enable_multiple_frontends: false

backend:
type: "vllm"
connector: null

prefill_environment: &worker-environment
VLLM_ENGINE_READY_TIMEOUT_S: "3600"
VLLM_FLOAT32_MATMUL_PRECISION: "high"
VLLM_FLASHINFER_ALLREDUCE_BACKEND: "mnnvl"
UCX_CUDA_IPC_ENABLE_MNNVL: "y"
UCX_MODULE_DIR: "/usr/local/lib/python3.12/dist-packages/nixl_cu13.libs/ucx"
UCX_RNDV_PIPELINE_ERROR_HANDLING: "y"
NCCL_CUMEM_ENABLE: "1"
NCCL_MNNVL_ENABLE: "1"
NCCL_NVLS_ENABLE: "1"

decode_environment: *worker-environment

vllm_config:
prefill:
no-enable-flashinfer-autotune: true
kv-transfer-config: '{"kv_connector": "NixlConnector", "kv_role": "kv_both"}'
attention-config: '{"backend": "FLASHINFER", "use_trtllm_attention": true, "indexer_kv_dtype": "fp8"}'
speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}'
tensor-parallel-size: 1
data-parallel-size: 2
data-parallel-rpc-port: 13345
enable-expert-parallel: true
trust-remote-code: true
no-enable-prefix-caching: true
block-size: 128
gpu-memory-utilization: 0.90
max-model-len: 9472
language-model-only: true
kv-cache-dtype: "fp8"
stream-interval: 32
max-cudagraph-capture-size: 2048
max-num-batched-tokens: 16384

decode:
no-enable-flashinfer-autotune: true
kv-transfer-config: '{"kv_connector": "NixlConnector", "kv_role": "kv_both"}'
attention-config: '{"backend": "FLASHINFER", "use_trtllm_attention": true, "indexer_kv_dtype": "fp8", "minimax_m3_msa_decode_backend": "cutlass"}'
speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}'
tensor-parallel-size: 4
enable-expert-parallel: false
trust-remote-code: true
no-enable-prefix-caching: true
block-size: 128
gpu-memory-utilization: 0.90
max-model-len: 9472
language-model-only: true
kv-cache-dtype: "fp8"
stream-interval: 32
max-num-seqs: 1024
max-num-batched-tokens: 16384
max-cudagraph-capture-size: 2048
cudagraph_mode: FULL_DECODE_ONLY

Check failure on line 104 in benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/mtp/1p1d-dep2-tp4-eagle3-c1-8k1k.yaml

View check run for this annotation

Claude / Claude Code Review

cudagraph_mode set as bare top-level key instead of via compilation-config JSON

All 11 new decode `vllm_config` blocks in this PR set `cudagraph_mode: FULL_DECODE_ONLY` as a bare top-level key (e.g. line 104 in `1p1d-dep2-tp4-eagle3-c1-8k1k.yaml`, and the same line in the other 10 sibling files), but every other recipe in this repo enables that mode by embedding it inside the `compilation-config` JSON blob (e.g. `compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY",...}'`). Since srt-slurm passes top-level `vllm_config` keys through as `--<key>` CLI flags, this likely

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🔴 All 11 new decode vllm_config blocks in this PR set cudagraph_mode: FULL_DECODE_ONLY as a bare top-level key (e.g. line 104 in 1p1d-dep2-tp4-eagle3-c1-8k1k.yaml, and the same line in the other 10 sibling files), but every other recipe in this repo enables that mode by embedding it inside the compilation-config JSON blob (e.g. compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY",...}'). Since srt-slurm passes top-level vllm_config keys through as --<key> CLI flags, this likely emits an unrecognized --cudagraph_mode flag (crashing decode server startup) or is silently dropped — either way defeating the PR's sole stated purpose of enabling FULL_DECODE_ONLY CUDA graphs on decode.

Extended reasoning...

The bug: every one of the 11 new decode vllm_config blocks added by this PR sets cudagraph_mode: FULL_DECODE_ONLY as a bare, top-level recipe key sitting alongside max-cudagraph-capture-size, kv-cache-dtype, etc. For example, in 1p1d-dep2-tp4-eagle3-c1-8k1k.yaml:

    decode:
      ...
      max-cudagraph-capture-size: 2048
      cudagraph_mode: FULL_DECODE_ONLY

Why this deviates from the established pattern: grepping the repo shows cudagraph_mode is enabled in 100+ places across dozens of recipe files, and in every single one of them it is embedded inside a compilation-config JSON string, never set as a bare key. For instance, the sibling minimax-m3/b200-fp4/8k1k/*.yaml recipes (same model family) do:

compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","custom_ops":["+rms_norm"],"pass_config":{}}'

The same pattern holds for the deepseek-v4 and kimi-k2.5-fp4 recipe families. This is because cudagraph_mode is a field of vLLM's CompilationConfig, not a standalone engine-args flag — there is no --cudagraph-mode/--cudagraph_mode CLI argument in vLLM's argparse; it can only be set via --compilation-config (JSON) or the -O dot-notation shorthand. That's precisely why every other recipe author routed it through the JSON blob instead of a bare key.

How this breaks at runtime: the other 20+ keys in these same vllm_config blocks (no-enable-flashinfer-autotune, kv-transfer-config, max-cudagraph-capture-size, tensor-parallel-size, etc.) are all kebab-case, matching srt-slurm's convention of passing top-level vllm_config keys through verbatim as --<key> CLI flags to vllm serve. cudagraph_mode is the only snake_case key in these files — it looks like the author copy-pasted the JSON field name but forgot to wrap it inside compilation-config. Under that passthrough convention, this key gets emitted as --cudagraph_mode FULL_DECODE_ONLY. Since vLLM has no such CLI flag, this either (a) is rejected by argparse as an unrecognized argument, crashing decode-server startup, or (b) is silently dropped by the recipe-to-CLI translation, in which case FULL_DECODE_ONLY is simply never applied.

Step-by-step proof:

  1. PR description states the sole functional change is: 'in all 11 decode vllm_config sections, add cudagraph_mode: FULL_DECODE_ONLY.'
  2. Every established recipe in the repo that has ever enabled this mode does so via compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY",...}' — confirmed by grep across 70+ files, including the same-model minimax-m3/b200-fp4 sibling recipes.
  3. These 11 new files instead add a bare top-level cudagraph_mode: FULL_DECODE_ONLY key at decode-block scope (e.g. line 104 of 1p1d-dep2-tp4-eagle3-c1-8k1k.yaml), with none of the 11 decode blocks touching compilation-config at all.
  4. srt-slurm's recipe schema maps top-level vllm_config keys to --<key> vLLM CLI flags 1:1 (evidenced by every other kebab-case key in the same block mapping directly to a real vLLM flag).
  5. vLLM's CLI has no top-level --cudagraph-mode/--cudagraph_mode engine argument — that field only exists inside CompilationConfig, settable via --compilation-config JSON.
  6. Therefore this key either crashes the decode vLLM server at startup (unrecognized argument) or is silently ignored, and in neither case is FULL_DECODE_ONLY CUDA graph mode actually enabled — defeating the entire stated purpose of the PR.

The fix: merge the key into the decode compilation-config JSON blob, consistent with every other recipe, e.g. add compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY"}' to each of the 11 decode blocks (or fold it into an existing compilation-config entry if one is later added) instead of the current bare cudagraph_mode: FULL_DECODE_ONLY key.


benchmark:
type: "sa-bench"
isl: 8192
osl: 1024
concurrencies: "1"
req_rate: "inf"
num_warmup_mult: 2
random_range_ratio: 0.8
use_chat_template: true
Original file line number Diff line number Diff line change
@@ -0,0 +1,114 @@
name: "minimax-m3-vllm-disagg-gb300-1p1d-dep2-tp4-mxfp8-8k1k-eagle3-c8"

model:
path: "minimax-m3-mxfp8"
container: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7"
precision: "fp8"

identity:
model:
repo: "MiniMaxAI/MiniMax-M3-MXFP8"
revision: "c5454eb03678d8710e54a4e0fc681b9f3b4a3dba"
container:
image: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7"
frameworks:
dynamo: "1.4.0.dev20260730"
vllm: "0.26.1rc1.dev255+g5e35a6f4f"

dynamo:
install: true
version: "1.4.0.dev20260730"
request_plane: "nats"

health_check:
max_attempts: 720
interval_seconds: 10

sbatch_directives:
mem: "0"
cpus-per-task: "72"

srun_options:
mem: "0"

resources:
gpu_type: "gb300"
gpus_per_node: 4
prefill_nodes: 1
decode_nodes: 1
prefill_workers: 1
decode_workers: 1
gpus_per_prefill: 2
gpus_per_decode: 4

frontend:
type: "dynamo"
enable_multiple_frontends: false

backend:
type: "vllm"
connector: null

prefill_environment: &worker-environment
VLLM_ENGINE_READY_TIMEOUT_S: "3600"
VLLM_FLOAT32_MATMUL_PRECISION: "high"
VLLM_FLASHINFER_ALLREDUCE_BACKEND: "mnnvl"
UCX_CUDA_IPC_ENABLE_MNNVL: "y"
UCX_MODULE_DIR: "/usr/local/lib/python3.12/dist-packages/nixl_cu13.libs/ucx"
UCX_RNDV_PIPELINE_ERROR_HANDLING: "y"
NCCL_CUMEM_ENABLE: "1"
NCCL_MNNVL_ENABLE: "1"
NCCL_NVLS_ENABLE: "1"

decode_environment: *worker-environment

vllm_config:
prefill:
no-enable-flashinfer-autotune: true
kv-transfer-config: '{"kv_connector": "NixlConnector", "kv_role": "kv_both"}'
attention-config: '{"backend": "FLASHINFER", "use_trtllm_attention": true, "indexer_kv_dtype": "fp8"}'
speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}'
tensor-parallel-size: 1
data-parallel-size: 2
data-parallel-rpc-port: 13345
enable-expert-parallel: true
trust-remote-code: true
no-enable-prefix-caching: true
block-size: 128
gpu-memory-utilization: 0.90
max-model-len: 9472
language-model-only: true
kv-cache-dtype: "fp8"
stream-interval: 32
max-cudagraph-capture-size: 2048
max-num-batched-tokens: 16384

decode:
no-enable-flashinfer-autotune: true
kv-transfer-config: '{"kv_connector": "NixlConnector", "kv_role": "kv_both"}'
attention-config: '{"backend": "FLASHINFER", "use_trtllm_attention": true, "indexer_kv_dtype": "fp8", "minimax_m3_msa_decode_backend": "cutlass"}'
speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}'
tensor-parallel-size: 4
enable-expert-parallel: false
trust-remote-code: true
no-enable-prefix-caching: true
block-size: 128
gpu-memory-utilization: 0.90
max-model-len: 9472
language-model-only: true
kv-cache-dtype: "fp8"
stream-interval: 32
max-num-seqs: 1024
max-num-batched-tokens: 16384
max-cudagraph-capture-size: 2048
cudagraph_mode: FULL_DECODE_ONLY

benchmark:
type: "sa-bench"
isl: 8192
osl: 1024
concurrencies: "8"
req_rate: "inf"
num_warmup_mult: 2
random_range_ratio: 0.8
use_chat_template: true
Loading
Loading