Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
19 commits
Select commit Hold shift + click to select a range
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
376 changes: 376 additions & 0 deletions .claude/agents/perf-test-sync.md

Large diffs are not rendered by default.

2 changes: 1 addition & 1 deletion jenkins/L0_Test.groovy
Original file line number Diff line number Diff line change
Expand Up @@ -3304,7 +3304,7 @@ def launchTestJobs(pipeline, testFilter)
"GB10-PyTorch-Post-Merge-1": ["gb10x-single", "l0_gb10", 1, 1],
// Disable GB300 stages due to nodes will be offline temporarily.
// "GB300-PyTorch-1": ["gb300-single", "l0_gb300", 1, 1],
// "GB300-4_GPUs-PyTorch-Post-Merge-1": ["gb300-x4", "l0_gb300_multi_gpus", 1, 1, 4],
"GB300-4_GPUs-PyTorch-Post-Merge-1": ["auto:gb300-x4", "l0_gb300_multi_gpus", 1, 1, 4],
// PerfSanity pre-merge tests
"GB200-4_GPUs-PyTorch-PerfSanity-1": ["auto:gb200-x4", "l0_gb200_multi_gpus_perf_sanity", 1, 1, 4],
// PerfSanity post-merge tests
Expand Down
109 changes: 109 additions & 0 deletions perf_test_sync_agg_report.html

Large diffs are not rendered by default.

7 changes: 0 additions & 7 deletions tests/integration/defs/perf/test_perf_sanity.py
Original file line number Diff line number Diff line change
Expand Up @@ -362,15 +362,8 @@ def to_match_keys(self) -> List[str]:
"l_cp",
"l_gpus_per_node",
"l_max_batch_size",
"b_disable_overlap_scheduler",
"b_enable_chunked_prefill",
"b_enable_attention_dp",
"b_enable_lm_head_tp_in_adp",
"s_serving_backend",
# attention_dp_config
"b_attention_dp_balance",
# cuda_graph_config
"b_enable_cuda_graph",
# kv_cache_config
"s_kv_cache_dtype",
# cache_transceiver_config
Expand Down
192 changes: 111 additions & 81 deletions tests/integration/test_lists/qa/llm_perf_multinode.txt

Large diffs are not rendered by default.

4 changes: 1 addition & 3 deletions tests/integration/test_lists/test-db/l0_a10.yml
Original file line number Diff line number Diff line change
Expand Up @@ -23,6 +23,7 @@ l0_a10:
- unittest/_torch/modeling/test_modeling_cohere2.py
- unittest/_torch/modeling/test_nemotron_nano_preprocessing.py
- unittest/_torch/modeling/test_modeling_parakeet.py
- unittest/_torch/modeling/test_modeling_radio.py
- unittest/_torch/sampler/test_trtllm_sampler.py
- unittest/_torch/executor/test_async_transfer_manager.py
- unittest/_torch/executor/test_scheduler_serializable_output.py
Expand Down Expand Up @@ -224,9 +225,6 @@ l0_a10:
stage: post_merge
backend: tensorrt
tests:
- test_e2e.py::test_mistral_e2e[use_py_session---]
- test_e2e.py::test_mistral_e2e[use_cpp_session-remove_input_padding--]
- test_e2e.py::test_mistral_e2e[use_py_session-remove_input_padding--]
- examples/test_bert.py::test_llm_bert_general[compare_hf-disable_remove_input_padding-disable_attention_plugin-disable_context_fmha-tp:1-pp:1-float32-BertModel-bert/bert-base-uncased]
- examples/test_bert.py::test_llm_bert_general[compare_hf-enable_remove_input_padding-use_attention_plugin-enable_context_fmha-tp:1-pp:1-float16-RobertaModel-bert/roberta-base]
- examples/test_bert.py::test_llm_bert_general[compare_hf-enable_remove_input_padding-disable_attention_plugin-disable_context_fmha-tp:1-pp:1-float16-RobertaForSequenceClassification-bert/twitter-roberta-base-emotion]
Expand Down
7 changes: 6 additions & 1 deletion tests/integration/test_lists/test-db/l0_b200.yml
Original file line number Diff line number Diff line change
Expand Up @@ -120,7 +120,12 @@ l0_b200:
# ------------- MoE: FlashInfer & TRTLLM symbol collision tests ---------------
- unittest/_torch/flashinfer/test_trtllm_flashinfer_symbol_collision.py
# --- MoE end
- unittest/_torch/multimodal
- unittest/_torch/multimodal/test_mm_encoder_standalone.py
- unittest/_torch/multimodal/test_multimodal_runtime.py
- unittest/_torch/multimodal/test_find_num_image_tokens.py
- unittest/_torch/multimodal/test_fuse_input_embeds.py
- unittest/_torch/multimodal/test_external_embedding.py
- unittest/_torch/multimodal/test_share_multiparams.py
- unittest/_torch/sampler
- unittest/_torch/speculative
- unittest/_torch/thop/parallel TIMEOUT (90)
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,25 @@
version: 0.0.1
l0_dgx_h200_perf_sanity:
- condition:
ranges:
system_gpu_count:
gte: 8
lte: 8
wildcards:
gpu:
- '*h200*'
linux_distribution_name: ubuntu*
cpu: x86_64
terms:
stage: post_merge
backend: pytorch
tests:


- perf/test_perf_sanity.py::test_e2e[disagg_upload-gen_only-h200_nemotron-super-fp8_8k1k_con64_ctx1_tp2_gen1_tp2_eplb0_mtp0_ccb-UCX]
- perf/test_perf_sanity.py::test_e2e[disagg_upload-gen_only-h200_qwen3-235b-a22b-fp8_8k1k_con512_ctx1_tp2_gen1_tep4_eplb0_mtp0_ccb-DEFAULT]
- perf/test_perf_sanity.py::test_e2e[disagg_upload-gen_only-h200_qwen3-32b-fp8_4k1k_con128_ctx1_tp1_gen1_tp2_eplb0_mtp0_ccb-DEFAULT]

# - perf/test_perf_sanity.py::test_e2e[disagg_upload-e2e-h200_nemotron-super-fp8_8k1k_con64_ctx1_tp2_gen1_tp2_eplb0_mtp0_ccb-UCX]
# - perf/test_perf_sanity.py::test_e2e[disagg_upload-e2e-h200_qwen3-235b-a22b-fp8_8k1k_con512_ctx1_tp2_gen1_tep4_eplb0_mtp0_ccb-DEFAULT]
# - perf/test_perf_sanity.py::test_e2e[disagg_upload-e2e-h200_qwen3-32b-fp8_4k1k_con128_ctx1_tp1_gen1_tp2_eplb0_mtp0_ccb-DEFAULT]
2 changes: 0 additions & 2 deletions tests/integration/test_lists/waives.txt
Original file line number Diff line number Diff line change
Expand Up @@ -364,8 +364,6 @@ accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_chunked_prefill[quant
accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_fp8_block_scales_cuda_graph_padding_4gpus[attention_dp=True-mtp_nextn=0] SKIP (https://nvbugs/6084447)
accuracy/test_llm_api_pytorch.py::TestQwen3_235B_A22B::test_nvfp4[latency_moe_trtllm_attention_dp] SKIP (https://nvbugs/6084568)
perf/test_perf_sanity.py::test_e2e[disagg_upload-gen_only-gb200_deepseek-r1-fp4_1k1k_con1024_ctx1_dep4_gen1_dep8_eplb0_mtp0_ccb-UCX] SKIP (https://nvbugs/6088149)
perf/test_perf_sanity.py::test_e2e[disagg_upload-gen_only-gb200_deepseek-r1-fp4_1k1k_con1024_ctx1_dep4_gen1_dep32_eplb0_mtp3_ccb-UCX] SKIP (https://nvbugs/6088149)
perf/test_perf_sanity.py::test_e2e[aggr_upload-k25_thinking_fp4_2_nodes_grace_blackwell-k25_thinking_fp4_dep8_32k8k] SKIP (https://nvbugs/6088149)
accuracy/test_llm_api_pytorch.py::TestNemotronNas::test_auto_dtype_tp8 SKIP (https://nvbugs/6070857)
accuracy/test_llm_api_autodeploy.py::TestNemotronNanoV3::test_accuracy[fp8-1-trtllm] SKIP (https://nvbugs/6094208)
accuracy/test_llm_api_autodeploy.py::TestNemotronNanoV3::test_accuracy[bf16-1-trtllm] SKIP (https://nvbugs/6094208)
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,91 @@
metadata:
model_name: super_fp8
precision: fp8
model_dir_name: NVIDIA-Nemotron-3-Super-120B-A12B-FP8
supported_gpus:
- H200
script_file: disaggr_torch.slurm
benchmark_type: 8k1k
# Native-target (Hopper) mirror of the Dynamo Nemotron-3-Super-FP8 TRT-LLM
# disagg deployment recipe:
# https://github.com/ai-dynamo/dynamo/tree/main/recipes/nemotron-3-super-fp8/trtllm/disagg
Comment on lines +1 to +11

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

⚠️ Potential issue | 🟡 Minor

Missing SPDX license header.

This file lacks the Apache 2.0 license header that is present in all other new YAML files added in this PR. Consider adding the standard header for consistency.

Proposed fix
+# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+# SPDX-License-Identifier: Apache-2.0
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
 metadata:
   model_name: super_fp8
📝 Committable suggestion

‼️ IMPORTANT
Carefully review the code before committing. Ensure that it accurately replaces the highlighted code, contains no missing lines, and has no issues with indentation. Thoroughly test & benchmark the code to ensure it meets the requirements.

Suggested change
metadata:
model_name: super_fp8
precision: fp8
model_dir_name: NVIDIA-Nemotron-3-Super-120B-A12B-FP8
supported_gpus:
- H200
script_file: disaggr_torch.slurm
benchmark_type: 8k1k
# Native-target (Hopper) mirror of the Dynamo Nemotron-3-Super-FP8 TRT-LLM
# disagg deployment recipe:
# https://github.com/ai-dynamo/dynamo/tree/main/recipes/nemotron-3-super-fp8/trtllm/disagg
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
metadata:
model_name: super_fp8
precision: fp8
model_dir_name: NVIDIA-Nemotron-3-Super-120B-A12B-FP8
supported_gpus:
- H200
script_file: disaggr_torch.slurm
benchmark_type: 8k1k
# Native-target (Hopper) mirror of the Dynamo Nemotron-3-Super-FP8 TRT-LLM
# disagg deployment recipe:
# https://github.com/ai-dynamo/dynamo/tree/main/recipes/nemotron-3-super-fp8/trtllm/disagg
🤖 Prompt for AI Agents
Verify each finding against the current code and only fix it if needed.

In
`@tests/scripts/perf-sanity/disaggregated/h200_nemotron-super-fp8_8k1k_con64_ctx1_tp2_gen1_tp2_eplb0_mtp0_ccb-UCX.yaml`
around lines 1 - 11, Add the standard Apache-2.0 SPDX license header to the top
of this YAML (above the existing metadata block) so it matches other new YAML
files in the PR; ensure the header exactly matches the project's SPDX/A2.0
header used elsewhere (e.g., "SPDX-License-Identifier: Apache-2.0" plus any
project comment lines) and place it before the metadata: key so the file begins
with the license header.

slurm:
script_file: disaggr_torch.slurm
partition: <partition>
account: <account>
job_time: 02:00:00
job_name: unified-benchmark
extra_args: "--gres=gpu:8"
numa_bind: true
benchmark:
mode: e2e
use_nv_sa_benchmark: false
multi_round: 10
benchmark_ratio: 0.0
streaming: true
concurrency_list: '64'
input_length: 8192
output_length: 1024
dataset_file: datasets/perf-ci/nemotron_super-8k1k-20480-ratio-1_for_serve.json
hardware:
gpus_per_node: 8
num_ctx_servers: 1
num_gen_servers: 1
environment:
container_mount: <container_mount>
container_image: <container_image>
model_path: <model_path>
trtllm_repo: ''
build_wheel: false
work_dir: <full_path_to_work_dir>
worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes"
server_env_var: "TRTLLM_SERVER_DISABLE_GC=1"
profiling:
nsys_on: false
accuracy:
enable_accuracy_test: false
worker_config:
gen:
print_iter_log: true
tensor_parallel_size: 2
moe_expert_parallel_size: 1
pipeline_parallel_size: 1
context_parallel_size: 1
enable_attention_dp: false
enable_chunked_prefill: true
max_batch_size: 16
max_num_tokens: 8192
trust_remote_code: true
cuda_graph_config:
enable_padding: true
max_batch_size: 16
kv_cache_config:
enable_block_reuse: false
free_gpu_memory_fraction: 0.85
moe_config:
backend: TRTLLM
cache_transceiver_config:
backend: UCX
disable_overlap_scheduler: false
ctx:
print_iter_log: true
tensor_parallel_size: 2
moe_expert_parallel_size: 1
pipeline_parallel_size: 1
context_parallel_size: 1
enable_attention_dp: false
enable_chunked_prefill: true
max_batch_size: 16
max_num_tokens: 8192
trust_remote_code: true
cuda_graph_config:
enable_padding: true
max_batch_size: 16
kv_cache_config:
enable_block_reuse: false
free_gpu_memory_fraction: 0.85
moe_config:
backend: TRTLLM
cache_transceiver_config:
backend: UCX
disable_overlap_scheduler: true
Original file line number Diff line number Diff line change
Expand Up @@ -3,30 +3,32 @@ metadata:
precision: fp8
model_dir_name: Qwen3-235B-A22B-FP8
supported_gpus:
- GB200
- GB300
- H200
script_file: disaggr_torch.slurm
benchmark_type: 1k1k
benchmark_type: 8k1k
# Native-target (Hopper) mirror of the Dynamo Qwen3-235B-A22B-FP8 TRT-LLM
# disagg deployment recipe:
# https://github.com/ai-dynamo/dynamo/tree/main/recipes/qwen3-235b-a22b-fp8/trtllm/disagg
slurm:
script_file: disaggr_torch.slurm
partition: <partition>
account: <account>
job_time: 02:00:00
job_name: unified-benchmark
extra_args: --gres=gpu:4
extra_args: "--gres=gpu:8"
numa_bind: true
benchmark:
mode: e2e
use_nv_sa_benchmark: true
multi_round: 8
benchmark_ratio: 0.8
use_nv_sa_benchmark: false
multi_round: 10
benchmark_ratio: 0.0
streaming: true
concurrency_list: '16'
input_length: 1024
concurrency_list: '512'
input_length: 8192
output_length: 1024
dataset_file: <dataset_file>
dataset_file: datasets/perf-ci/qwen3_235b-8k1k-20480-ratio-1_for_serve.json
hardware:
gpus_per_node: 4
gpus_per_node: 8
num_ctx_servers: 1
num_gen_servers: 1
environment:
Expand All @@ -36,55 +38,60 @@ environment:
trtllm_repo: ''
build_wheel: false
work_dir: <full_path_to_work_dir>
worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 MIMALLOC_PURGE_DELAY=0 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes"
server_env_var: TRTLLM_SERVER_DISABLE_GC=1
worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes"
server_env_var: "TRTLLM_SERVER_DISABLE_GC=1"
profiling:
nsys_on: false
accuracy:
enable_accuracy_test: false
worker_config:
gen:
print_iter_log: true
tensor_parallel_size: 4
moe_expert_parallel_size: 4
enable_attention_dp: false
moe_tensor_parallel_size: 1
pipeline_parallel_size: 1
max_batch_size: 64
max_num_tokens: 2048
max_seq_len: 2051
context_parallel_size: 1
enable_attention_dp: false
enable_chunked_prefill: false
trust_remote_code: true
max_batch_size: 512
max_num_tokens: 1024
max_seq_len: 8192
cuda_graph_config:
enable_padding: true
max_batch_size: 128
print_iter_log: true
max_batch_size: 512
kv_cache_config:
enable_block_reuse: true
free_gpu_memory_fraction: 0.7
enable_block_reuse: false
free_gpu_memory_fraction: 0.95
dtype: fp8
moe_config:
backend: TRTLLM
backend: DEEPGEMM
cache_transceiver_config:
max_tokens_in_buffer: 2048
backend: NIXL
stream_interval: 20
num_postprocess_workers: 4
allreduce_strategy: MNNVL
backend: DEFAULT
disable_overlap_scheduler: false
ctx:
max_batch_size: 32
max_num_tokens: 2048
max_seq_len: 2051
tensor_parallel_size: 4
moe_expert_parallel_size: 4
enable_attention_dp: false
pipeline_parallel_size: 1
print_iter_log: true
cuda_graph_config: null
disable_overlap_scheduler: true
tensor_parallel_size: 2
moe_expert_parallel_size: 1
moe_tensor_parallel_size: 2
pipeline_parallel_size: 1
context_parallel_size: 1
enable_attention_dp: false
enable_chunked_prefill: false
trust_remote_code: true
max_batch_size: 2
max_num_tokens: 8192
max_seq_len: 8192
cuda_graph_config:
enable_padding: true
max_batch_size: 2
kv_cache_config:
enable_block_reuse: true
free_gpu_memory_fraction: 0.7
dtype: fp8
moe_config:
backend: TRTLLM
backend: DEEPGEMM
cache_transceiver_config:
max_tokens_in_buffer: 2048
backend: NIXL
backend: DEFAULT
disable_overlap_scheduler: true
Loading
Loading