Skip to content
Merged
Show file tree
Hide file tree
Changes from 4 commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .github/configs/nvidia-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -9014,7 +9014,7 @@ kimik2.5-fp4-gb200-dynamo-vllm:
dp-attn: true

dsv4-fp4-b200-dynamo-vllm:
image: vllm/vllm-openai:v0.20.1
image: vllm/vllm-openai:v0.23.0
model: deepseek-ai/DeepSeek-V4-Pro
model-prefix: dsv4
runner: b200-multinode
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -17,7 +17,7 @@ name: "svf-vllm-disagg-b200-high-tpt-megamoe"
# absorb cold-cache model loads.
model:
path: "deepseek-v4-pro"
container: "vllm/vllm-openai:v0.20.1"
container: "vllm/vllm-openai:v0.23.0"
precision: "fp4"

dynamo:
Expand Down Expand Up @@ -83,13 +83,13 @@ backend:
enforce-eager: true
max-model-len: 9280
max-num-seqs: 16
max-num-batched-tokens: 32768
max-num-batched-tokens: 16384
trust-remote-code: true
no-enable-prefix-caching: true
no-enable-flashinfer-autotune: true
no-async-scheduling: true
block-size: 256
gpu-memory-utilization: 0.95
gpu-memory-utilization: 0.9
no-disable-hybrid-kv-cache-manager: true
enable-sleep-mode: true
numa-bind: true
Expand Down Expand Up @@ -132,7 +132,7 @@ identity:
repo: "deepseek-ai/DeepSeek-V4-Pro"
revision: "0366e4e064385807ea86b088a5c6c878ff23343b"
container:
image: "vllm/vllm-openai:v0.20.1"
image: "vllm/vllm-openai:v0.23.0"
frameworks:
dynamo: "1.2.0.dev20260426"
vllm: "0.20.0"
vllm: "0.23.0"
Original file line number Diff line number Diff line change
Expand Up @@ -17,7 +17,7 @@ name: "svf-vllm-disagg-b200-low-latency"
# absorb cold-cache model loads.
model:
path: "deepseek-v4-pro"
container: "vllm/vllm-openai:v0.20.1"
container: "vllm/vllm-openai:v0.23.0"
precision: "fp4"

dynamo:
Expand Down Expand Up @@ -131,7 +131,7 @@ benchmark:

identity:
container:
image: "vllm/vllm-openai:v0.20.1"
image: "vllm/vllm-openai:v0.23.0"
frameworks:
dynamo: "1.2.0.dev20260426"
vllm: "0.20.0"
vllm: "0.23.0"
Original file line number Diff line number Diff line change
Expand Up @@ -17,7 +17,7 @@ name: "svf-vllm-disagg-b200-low-middle-curve"
# absorb cold-cache model loads.
model:
path: "deepseek-v4-pro"
container: "vllm/vllm-openai:v0.20.1"
container: "vllm/vllm-openai:v0.23.0"
precision: "fp4"

dynamo:
Expand Down Expand Up @@ -132,7 +132,7 @@ benchmark:

identity:
container:
image: "vllm/vllm-openai:v0.20.1"
image: "vllm/vllm-openai:v0.23.0"
frameworks:
dynamo: "1.2.0.dev20260426"
vllm: "0.20.0"
vllm: "0.23.0"
Original file line number Diff line number Diff line change
Expand Up @@ -17,7 +17,7 @@ name: "svf-vllm-disagg-b200-max-tpt-megamoe"
# absorb cold-cache model loads.
model:
path: "deepseek-v4-pro"
container: "vllm/vllm-openai:v0.20.1"
container: "vllm/vllm-openai:v0.23.0"
precision: "fp4"

dynamo:
Expand Down Expand Up @@ -83,13 +83,13 @@ backend:
enforce-eager: true
max-model-len: 9280
max-num-seqs: 16
max-num-batched-tokens: 32768
max-num-batched-tokens: 16384
trust-remote-code: true
no-enable-prefix-caching: true
no-enable-flashinfer-autotune: true
no-async-scheduling: true
block-size: 256
gpu-memory-utilization: 0.95
gpu-memory-utilization: 0.9
no-disable-hybrid-kv-cache-manager: true
enable-sleep-mode: true
numa-bind: true
Expand Down Expand Up @@ -132,7 +132,7 @@ identity:
repo: "deepseek-ai/DeepSeek-V4-Pro"
revision: "0366e4e064385807ea86b088a5c6c878ff23343b"
container:
image: "vllm/vllm-openai:v0.20.1"
image: "vllm/vllm-openai:v0.23.0"
frameworks:
dynamo: "1.2.0.dev20260426"
vllm: "0.20.0"
vllm: "0.23.0"
Original file line number Diff line number Diff line change
Expand Up @@ -17,7 +17,7 @@ name: "svf-vllm-disagg-b200-mid-curve-megamoe"
# absorb cold-cache model loads.
model:
path: "deepseek-v4-pro"
container: "vllm/vllm-openai:v0.20.1"
container: "vllm/vllm-openai:v0.23.0"
precision: "fp4"

dynamo:
Expand Down Expand Up @@ -83,13 +83,13 @@ backend:
enforce-eager: true
max-model-len: 9280
max-num-seqs: 16
max-num-batched-tokens: 32768
max-num-batched-tokens: 16384
trust-remote-code: true
no-enable-prefix-caching: true
no-enable-flashinfer-autotune: true
no-async-scheduling: true
block-size: 256
gpu-memory-utilization: 0.95
gpu-memory-utilization: 0.9
no-disable-hybrid-kv-cache-manager: true
enable-sleep-mode: true
numa-bind: true
Expand Down Expand Up @@ -132,7 +132,7 @@ identity:
repo: "deepseek-ai/DeepSeek-V4-Pro"
revision: "0366e4e064385807ea86b088a5c6c878ff23343b"
container:
image: "vllm/vllm-openai:v0.20.1"
image: "vllm/vllm-openai:v0.23.0"
frameworks:
dynamo: "1.2.0.dev20260426"
vllm: "0.20.0"
vllm: "0.23.0"
7 changes: 7 additions & 0 deletions perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -4153,3 +4153,10 @@
- "Run the PR #1891 MiniMax-M3 MXFP8 B300 Dynamo-vLLM recipe set on top of current main."
- "Uses the vllm/vllm-openai:minimax-m3-0618-x86_64-cu130 image and the TEP4/TEP8 8k1k topologies not covered by PR #1890."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/1891

- config-keys:
- dsv4-fp4-b200-dynamo-vllm
description:
- "Update the DeepSeek-V4-Pro B200 disaggregated Dynamo-vLLM benchmark to the vllm/vllm-openai:v0.23.0 image"
- "Lower max-num-batched-tokens to 16384 and gpu-memory-utilization to 0.9 on the high-throughput and max-throughput recipes to avoid OOM"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/1899