From 9b1aeeb3aa2fa85a1475dc31b90dd2bfd1ade73c Mon Sep 17 00:00:00 2001 From: Chenfei Zhang Date: Fri, 27 Feb 2026 02:05:10 -0800 Subject: [PATCH 01/12] update Signed-off-by: Chenfei Zhang --- jenkins/L0_MergeRequest.groovy | 20 +- jenkins/L0_Test.groovy | 15 +- jenkins/scripts/perf/README.md | 150 +- .../perf/disaggregated/slurm_launch_draft.sh | 4 +- jenkins/scripts/perf/disaggregated/submit.py | 65 +- jenkins/scripts/perf/get_pre_merge_html.py | 276 +++ jenkins/scripts/perf/local/README.md | 153 +- jenkins/scripts/perf/local/submit.py | 131 +- jenkins/scripts/perf/perf_regression.py | 275 --- jenkins/scripts/perf/perf_utils.py | 1620 +++++++++++++++++ tensorrt_llm/_torch/pyexecutor/py_executor.py | 10 +- .../defs/perf/open_search_db_utils.py | 313 ++-- .../integration/defs/perf/test_perf_sanity.py | 50 +- ...sanity_ctx1_node1_gpu1_gen1_node1_gpu2.yml | 2 +- ...sanity_ctx1_node1_gpu1_gen1_node1_gpu4.yml | 2 +- tests/integration/test_lists/waives.txt | 19 +- tests/scripts/perf-sanity/README.md | 25 +- .../config_database_b200_nvl.yaml | 0 .../config_database_h200_sxm.yaml | 0 ...eek_r1_fp4_v2_2_nodes_grace_blackwell.yaml | 0 .../deepseek_r1_fp4_v2_blackwell.yaml | 0 .../deepseek_r1_fp4_v2_grace_blackwell.yaml | 0 .../deepseek_r1_fp8_blackwell.yaml | 0 .../deepseek_v32_fp4_blackwell.yaml | 0 .../deepseek_v32_fp4_grace_blackwell.yaml | 0 ...eek_r1_fp4_v2_2_nodes_grace_blackwell.yaml | 0 .../gpt_oss_120b_fp4_blackwell.yaml | 0 .../gpt_oss_120b_fp4_grace_blackwell.yaml | 0 ..._thinking_fp4_2_nodes_grace_blackwell.yaml | 0 .../k2_thinking_fp4_blackwell.yaml | 0 .../k2_thinking_fp4_grace_blackwell.yaml | 0 ...tx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml | 0 ...tx1_dep4_gen1_dep8_eplb0_mtp1_ccb-UCX.yaml | 0 ...tx1_dep4_gen1_dep8_eplb0_mtp3_ccb-UCX.yaml | 0 ...tx1_dep4_gen1_dep8_eplb0_mtp1_ccb-UCX.yaml | 0 ...tx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml | 0 ...tx1_dep4_gen1_dep8_eplb0_mtp1_ccb-UCX.yaml | 0 ...tx1_pp8_gen1_dep16_eplb0_mtp2_ccb-UCX.yaml | 0 ...ctx1_pp8_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml | 0 ...tx1_pp8_gen1_dep32_eplb0_mtp3_ccb-UCX.yaml | 0 ...x1_dep4_gen1_dep32_eplb0_mtp3_ccb-UCX.yaml | 0 ...tx1_dep4_gen1_dep8_eplb0_mtp0_ccb-UCX.yaml | 0 ...tx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml | 0 ...tx1_dep4_gen1_dep4_eplb0_mtp1_ccb-UCX.yaml | 0 ...x1_dep4_gen1_dep32_eplb0_mtp3_ccb-UCX.yaml | 0 ...tx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml | 0 ...x1_dep4_gen1_dep16_eplb0_mtp1_ccb-UCX.yaml | 0 ..._dep4_gen1_dep32_eplb256_mtp3_ccb-UCX.yaml | 0 ...tx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml | 0 ...tx1_dep4_gen1_dep4_eplb0_mtp1_ccb-UCX.yaml | 0 ...tx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml | 0 ..._dep4_gen1_dep32_eplb288_mtp1_ccb-UCX.yaml | 0 ...x1_dep4_gen1_dep32_eplb0_mtp3_ccb-UCX.yaml | 0 ..._dep4_gen1_dep32_eplb256_mtp3_ccb-UCX.yaml | 0 ...tx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml | 0 ..._dep4_gen1_dep32_eplb256_mtp0_ccb-UCX.yaml | 0 ...ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-UCX.yaml | 0 ...ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-UCX.yaml | 0 ..._ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-UCX.yaml | 0 ..._ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-UCX.yaml | 0 ..._ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-UCX.yaml | 0 ...ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-UCX.yaml | 0 ..._dep4_gen1_dep32_eplb384_mtp0_ccb-UCX.yaml | 0 ...tx1_dep4_gen1_dep8_eplb0_mtp0_ccb-UCX.yaml | 0 ...tx1_dep4_gen1_tep4_eplb0_mtp0_ccb-UCX.yaml | 0 ..._dep4_gen1_dep32_eplb416_mtp3_ccb-UCX.yaml | 0 ..._dep4_gen1_dep16_eplb384_mtp0_ccb-UCX.yaml | 0 ...tx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml | 0 ...ctx1_tp1_gen1_dep8_eplb0_mtp0_ccb-UCX.yaml | 0 ...ctx1_tp1_gen1_tep4_eplb0_mtp0_ccb-UCX.yaml | 0 ...x1_dep4_gen1_dep32_eplb0_mtp3_ccb-UCX.yaml | 0 tests/test_common/error_utils.py | 13 +- 72 files changed, 2503 insertions(+), 640 deletions(-) create mode 100644 jenkins/scripts/perf/get_pre_merge_html.py delete mode 100644 jenkins/scripts/perf/perf_regression.py create mode 100644 jenkins/scripts/perf/perf_utils.py rename tests/scripts/perf-sanity/{ => aggregated}/config_database_b200_nvl.yaml (100%) rename tests/scripts/perf-sanity/{ => aggregated}/config_database_h200_sxm.yaml (100%) rename tests/scripts/perf-sanity/{ => aggregated}/deepseek_r1_fp4_v2_2_nodes_grace_blackwell.yaml (100%) rename tests/scripts/perf-sanity/{ => aggregated}/deepseek_r1_fp4_v2_blackwell.yaml (100%) rename tests/scripts/perf-sanity/{ => aggregated}/deepseek_r1_fp4_v2_grace_blackwell.yaml (100%) rename tests/scripts/perf-sanity/{ => aggregated}/deepseek_r1_fp8_blackwell.yaml (100%) rename tests/scripts/perf-sanity/{ => aggregated}/deepseek_v32_fp4_blackwell.yaml (100%) rename tests/scripts/perf-sanity/{ => aggregated}/deepseek_v32_fp4_grace_blackwell.yaml (100%) rename tests/scripts/perf-sanity/{ => aggregated}/gb300_deepseek_r1_fp4_v2_2_nodes_grace_blackwell.yaml (100%) rename tests/scripts/perf-sanity/{ => aggregated}/gpt_oss_120b_fp4_blackwell.yaml (100%) rename tests/scripts/perf-sanity/{ => aggregated}/gpt_oss_120b_fp4_grace_blackwell.yaml (100%) rename tests/scripts/perf-sanity/{ => aggregated}/k2_thinking_fp4_2_nodes_grace_blackwell.yaml (100%) rename tests/scripts/perf-sanity/{ => aggregated}/k2_thinking_fp4_blackwell.yaml (100%) rename tests/scripts/perf-sanity/{ => aggregated}/k2_thinking_fp4_grace_blackwell.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/b200_deepseek-r1-fp4_1k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/b200_deepseek-r1-fp4_1k1k_con2048_ctx1_dep4_gen1_dep8_eplb0_mtp1_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/b200_deepseek-r1-fp4_1k1k_con256_ctx1_dep4_gen1_dep8_eplb0_mtp3_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/b200_deepseek-r1-fp4_8k1k_con1536_ctx1_dep4_gen1_dep8_eplb0_mtp1_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/b200_deepseek-r1-fp4_8k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/b200_deepseek-r1-fp4_8k1k_con256_ctx1_dep4_gen1_dep8_eplb0_mtp1_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_deepseek-r1-fp4_128k8k_con128_ctx1_pp8_gen1_dep16_eplb0_mtp2_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_deepseek-r1-fp4_128k8k_con1_ctx1_pp8_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_deepseek-r1-fp4_128k8k_con64_ctx1_pp8_gen1_dep32_eplb0_mtp3_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_deepseek-r1-fp4_1k1k_con1024_ctx1_dep4_gen1_dep32_eplb0_mtp3_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_deepseek-r1-fp4_1k1k_con1024_ctx1_dep4_gen1_dep8_eplb0_mtp0_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_deepseek-r1-fp4_1k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_deepseek-r1-fp4_1k1k_con3072_ctx1_dep4_gen1_dep4_eplb0_mtp1_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_deepseek-r1-fp4_8k1k_con1024_ctx1_dep4_gen1_dep32_eplb0_mtp3_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_deepseek-r1-fp4_8k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_deepseek-r1-fp4_8k1k_con4096_ctx1_dep4_gen1_dep16_eplb0_mtp1_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_deepseek-v32-fp4_1k1k_con1024_ctx1_dep4_gen1_dep32_eplb256_mtp3_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_deepseek-v32-fp4_1k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_deepseek-v32-fp4_1k1k_con2048_ctx1_dep4_gen1_dep4_eplb0_mtp1_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_deepseek-v32-fp4_32k4k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_deepseek-v32-fp4_32k4k_con2048_ctx1_dep4_gen1_dep32_eplb288_mtp1_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_deepseek-v32-fp4_32k4k_con256_ctx1_dep4_gen1_dep32_eplb0_mtp3_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_deepseek-v32-fp4_8k1k_con1024_ctx1_dep4_gen1_dep32_eplb256_mtp3_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_deepseek-v32-fp4_8k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_deepseek-v32-fp4_8k1k_con4096_ctx1_dep4_gen1_dep32_eplb256_mtp0_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_gpt-oss-120b-fp4_1k1k_con2048_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_gpt-oss-120b-fp4_1k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_gpt-oss-120b-fp4_1k1k_con64_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_gpt-oss-120b-fp4_8k1k_con128_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_gpt-oss-120b-fp4_8k1k_con4_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_gpt-oss-120b-fp4_8k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_kimi-k2-thinking-fp4_1k1k_con2048_ctx1_dep4_gen1_dep32_eplb384_mtp0_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_kimi-k2-thinking-fp4_1k1k_con4096_ctx1_dep4_gen1_dep8_eplb0_mtp0_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_kimi-k2-thinking-fp4_1k1k_con4_ctx1_dep4_gen1_tep4_eplb0_mtp0_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_kimi-k2-thinking-fp4_8k1k_con1024_ctx1_dep4_gen1_dep32_eplb416_mtp3_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_kimi-k2-thinking-fp4_8k1k_con4096_ctx1_dep4_gen1_dep16_eplb384_mtp0_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_kimi-k2-thinking-fp4_8k1k_con4_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_qwen3-235b-fp4_8k1k_con1024_ctx1_tp1_gen1_dep8_eplb0_mtp0_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb200_qwen3-235b-fp4_8k1k_con64_ctx1_tp1_gen1_tep4_eplb0_mtp0_ccb-UCX.yaml (100%) rename tests/{integration/defs/perf/disagg/test_configs/disagg/perf-sanity => scripts/perf-sanity/disaggregated}/gb300_deepseek-r1-fp4_1k1k_con1024_ctx1_dep4_gen1_dep32_eplb0_mtp3_ccb-UCX.yaml (100%) diff --git a/jenkins/L0_MergeRequest.groovy b/jenkins/L0_MergeRequest.groovy index cd75b768c799..af271d4413fb 100644 --- a/jenkins/L0_MergeRequest.groovy +++ b/jenkins/L0_MergeRequest.groovy @@ -867,30 +867,30 @@ def collectTestResults(pipeline, testFilter) junit(testResults: '**/results*.xml', allowEmptyResults : true) } // Collect test result stage - stage("Collect Perf Regression Result") { + stage("Collect Perf Sanity Test Result") { def yamlFiles = sh( returnStdout: true, - script: 'find . -type f -name "regression_data.yaml" 2>/dev/null || true' + script: 'find . -type f -name "perf_data.yaml" 2>/dev/null || true' ).trim() - echo "Regression data yaml files: ${yamlFiles}" + echo "Perf data yaml files: ${yamlFiles}" if (yamlFiles) { def yamlFileList = yamlFiles.split(/\s+/).collect { it.trim() }.findAll { it }.join(",") - echo "Found regression data files: ${yamlFileList}" + echo "Found perf data files: ${yamlFileList}" trtllm_utils.llmExecStepWithRetry(pipeline, script: "apk add python3") trtllm_utils.llmExecStepWithRetry(pipeline, script: "apk add py3-pip") trtllm_utils.llmExecStepWithRetry(pipeline, script: "pip3 config set global.break-system-packages true") trtllm_utils.llmExecStepWithRetry(pipeline, script: "pip3 install pyyaml") sh """ - python3 llm/jenkins/scripts/perf/perf_regression.py \ + python3 llm/jenkins/scripts/perf/get_pre_merge_html.py \ --input-files=${yamlFileList} \ - --output-file=perf_regression.html + --output-file=perf_sanity_report.html """ - trtllm_utils.uploadArtifacts("perf_regression.html", "${UPLOAD_PATH}/test-results/") - echo "Perf regression report: https://urm.nvidia.com/artifactory/${UPLOAD_PATH}/test-results/perf_regression.html" + trtllm_utils.uploadArtifacts("perf_sanity_report.html", "${UPLOAD_PATH}/test-results/") + echo "Perf sanity report: https://urm.nvidia.com/artifactory/${UPLOAD_PATH}/test-results/perf_sanity_report.html" } else { - echo "No regression_data.yaml files found." + echo "No perf_data.yaml files found." } - } // Collect Perf Regression Result stage + } // Collect Perf Sanity Test Result stage stage("Rerun Report") { sh "rm -rf rerun && mkdir -p rerun" sh "find . -type f -wholename '*/rerun_results.xml' -exec sh -c 'mv \"{}\" \"rerun/\$(basename \$(dirname \"{}\"))_rerun_results.xml\"' \\; || true" diff --git a/jenkins/L0_Test.groovy b/jenkins/L0_Test.groovy index 19799dcecf37..3cc7b5453153 100644 --- a/jenkins/L0_Test.groovy +++ b/jenkins/L0_Test.groovy @@ -1216,7 +1216,8 @@ def runLLMTestlistWithSbatch(pipeline, platform, testList, config=VANILLA_CONFIG --run-sh ${scriptRunPathNode} \\ --install-sh ${scriptInstallPathNode} \\ --script-prefix ${scriptLaunchPrefixPathLocal} \\ - --srun-args ${scriptLaunchSrunArgsPathLocal} + --srun-args ${scriptLaunchSrunArgsPathLocal} \\ + --split-group ${splitId} """ } else { if(nodeCount > 1) { @@ -3393,7 +3394,7 @@ def launchTestJobs(pipeline, testFilter) "GB200-8_GPUs-2_Nodes-PyTorch-Disagg-PerfSanity-CTX1-NODE1-GPU1-GEN1-NODE1-GPU2-Post-Merge", "auto:gb200-flex", "l0_gb200_multi_nodes_perf_sanity_ctx1_node1_gpu1_gen1_node1_gpu2", - 3, + 2, 8, 2 ) @@ -3401,7 +3402,7 @@ def launchTestJobs(pipeline, testFilter) "GB200-8_GPUs-2_Nodes-PyTorch-Disagg-PerfSanity-CTX1-NODE1-GPU1-GEN1-NODE1-GPU4-Post-Merge", "auto:gb200-flex", "l0_gb200_multi_nodes_perf_sanity_ctx1_node1_gpu1_gen1_node1_gpu4", - 4, + 3, 8, 2 ) @@ -3414,14 +3415,6 @@ def launchTestJobs(pipeline, testFilter) 2 ) // 3 Nodes - multiNodesSBSAConfigs += buildStageConfigs( - "GB200-12_GPUs-3_Nodes-PyTorch-Disagg-PerfSanity-CTX1-NODE1-GPU1-GEN1-NODE2-GPU8-Post-Merge", - "auto:gb200-flex", - "l0_gb200_multi_nodes_perf_sanity_ctx1_node1_gpu1_gen1_node2_gpu8", - 1, - 12, - 3 - ) multiNodesSBSAConfigs += buildStageConfigs( "GB200-12_GPUs-3_Nodes-PyTorch-Disagg-PerfSanity-CTX1-NODE1-GPU4-GEN1-NODE2-GPU8-Post-Merge", "auto:gb200-flex", diff --git a/jenkins/scripts/perf/README.md b/jenkins/scripts/perf/README.md index 68209344570c..625862ab9dea 100644 --- a/jenkins/scripts/perf/README.md +++ b/jenkins/scripts/perf/README.md @@ -1,64 +1,142 @@ -# Perf Sanity Triage +# Perf Sanity Scripts -This directory contains `perf_sanity_triage.py`, a helper script for querying -and updating perf sanity data in OpenSearch, and for sending regression -summaries to Slack. +This directory contains scripts for running perf sanity tests and managing perf sanity data. -## Basic Usage +## Directory Structure -This script is run by the Jenkins pipeline. Inputs are configured in `jenkins/runPerfSanityTriage.groovy`: +``` +jenkins/scripts/perf/ + aggregated/ + slurm_launch_draft.sh # Draft template for aggregated SLURM launch scripts + disaggregated/ + submit.py # CI pipeline submit script (disaggregated only) + slurm_launch_draft.sh # Draft template for disaggregated SLURM launch scripts + local/ + submit.py # Local submit script (aggregated and disaggregated) + slurm_install.sh # Build wheel + pip install inside container + slurm_run.sh # Run pytest inside container + perf_utils.py # Shared utilities (regression detection, baseline, charts, OpenSearch queries) + get_pre_merge_html.py # Pre-merge HTML report with history, baseline, and threshold + perf_sanity_triage.py # Query/update OpenSearch data and send Slack notifications +``` + +## Submit Scripts + +Both `local/submit.py` and `disaggregated/submit.py` share a similar workflow. They read +a test config YAML and use the appropriate draft template +(`aggregated/slurm_launch_draft.sh` or `disaggregated/slurm_launch_draft.sh`) to generate +a complete `slurm_launch.sh`. Then the user or CI pipeline can run `sbatch slurm_launch.sh` +to submit the job. Inside the SLURM job, `slurm_install.sh` builds the wheel and runs +installation, then `slurm_run.sh` runs pytest. + +``` +submit.py + | + v +slurm_launch.sh (generated) + | + |-- srun --> slurm_install.sh (build wheel + pip install) + |-- srun --> slurm_run.sh (run pytest) +``` + +Both submit scripts read `AGG_CONFIG_FOLDER` and `DISAGG_CONFIG_FOLDER` environment +variables (with defaults of `tests/scripts/perf-sanity/aggregated` and +`tests/scripts/perf-sanity/disaggregated`) and propagate them via `PYTEST_COMMON_VARS` +into the pytest execution environment where `test_perf_sanity.py` uses them to locate +config files. + +### `local/submit.py` + +Used for **local runs**. Supports both **aggregated** and **disaggregated** modes. It +detects the mode from the test config YAML (aggregated configs have `server_configs`, +disaggregated configs have `worker_config`) and selects the correct draft template +automatically. + +See [`local/README.md`](local/README.md) for full argument reference and examples. + +### `disaggregated/submit.py` + +Used by the **CI pipeline** (called from `jenkins/L0_Test.groovy`'s +`runLLMTestlistWithSbatch`). Only supports **disaggregated** mode. It receives a +script prefix and srun args from the CI pipeline and combines them with disagg-specific +environment variables and hardware configuration to generate `slurm_launch.sh`. + +## Shared Utilities + +### `perf_utils.py` + +Shared module imported by `get_post_merge_html.py`, `get_pre_merge_html.py`, and +`perf_sanity_triage.py`. Contains: -- `BRANCH`: repo branch to checkout -- `OPEN_SEARCH_PROJECT_NAME`: OpenSearch project name -- `OPERATION`: operation to perform (see Operations below) -- `QUERY_JOB_NUMBER`: number of latest jobs to query (OPERATION = "SLACK BOT SENDS MESSAGE" only) -- `SLACK_CHANNEL_ID`: Slack channel IDs (OPERATION = "SLACK BOT SENDS MESSAGE" only) -- `SLACK_BOT_TOKEN`: Slack bot token (OPERATION = "SLACK BOT SENDS MESSAGE" only) +- **Constants**: `CHART_METRICS` (4 key throughput metrics), `METRIC_LABELS`, + algorithm parameters, curve type colors/labels. +- **Baseline computation**: Rolling smooth (window=3) + P95 percentile algorithm. + Replaces the previous `max(daily_values)` approach which was vulnerable to + occasional spikes inflating the baseline. +- **Regression detection**: Two-step classification (regression check + subtype + pattern matching). Supports per-metric thresholds from baseline data + (`d_threshold_pre_merge_*` fields, defaulting to 5%). +- **OpenSearch query + grouping**: `get_history_data()` queries both baseline and + non-baseline data, groups by `(s_test_case_name, s_gpu_type)`. +- **SVG chart generation**: Unified chart function supporting history lines, + new data points, baseline line, threshold line, curve type badges, and jump + interval shading. +- **HTML dashboard**: `generate_post_merge_html()` produces a full interactive + report with three-way cascading filters and click-to-inspect data-point popups. -## Operations +## Post-Processing and Triage -### 1) `SLACK BOT SENDS MESSAGE` +### `get_pre_merge_html.py` -Queries regression data (post-merge only) and sends a formatted summary to -Slack. The query filters for: +Triggered at the end of the CI pipeline in `jenkins/L0_MergeRequest.groovy`. It has +3 main functions: -- `b_is_valid = true` -- `b_is_post_merge = true` -- `b_is_regression = true` -- `b_is_baseline = false` +1. **`load_perf_data`**: Reads perf_data.yaml files produced by test stages and + gathers all new perf data together. +2. **`get_pre_merge_history_data`**: Queries OpenSearch for post-merge history data + (both baseline and non-baseline), grouped by `(s_test_case_name, s_gpu_type)`. +3. **`generate_pre_merge_html`**: Generates an HTML report visualizing each test + case's key metrics (`d_seq_throughput`, `d_token_throughput`, + `d_total_token_throughput`, `d_user_throughput`) with history curve, new data + points, baseline line, and threshold line for regression comparison. -**Format** +### `perf_sanity_triage.py` + +Triggered by `jenkins/runPerfSanityTriage.groovy`. It supports two operations: + +1. **`SLACK BOT SENDS MESSAGE`**: Runs the perf-regression-detector pipeline + (`get_history_data` -> `get_baseline` -> `classify_test_case` -> + `generate_post_merge_html`), then sends the generated HTML dashboard to a + Slack channel. + +2. **`UPDATE SET ... (WHERE ...)`**: Updates fields on existing perf records that match + a query scope and posts the updated documents back to OpenSearch. + +**Examples** ``` SLACK BOT SENDS MESSAGE ``` -### 2) `UPDATE SET ... (WHERE ...)` +``` +UPDATE SET b_is_valid=false WHERE s_test_case_name='test1' +UPDATE SET b_is_valid=false WHERE ts_created <= 'Feb 18, 2026 @ 22:32:02.960' AND s_test_case_name='test1' +``` -Updates fields on existing perf records that match a query scope and posts the -updated documents back to OpenSearch. +See the `UPDATE` operation section below for supported operators and date formats. -**Operators** +#### UPDATE Operators - SET clause: Only `=` is supported. - WHERE clause: Supports `=`, `!=`, `>`, `<`, `>=`, `<=` operators. - `=` and `!=` operators are allowed for all fields. - `>`, `<`, `>=`, `<=` operators are only allowed for `ts_created` field (timestamp) or fields starting with `d_` (double type) or `l_` (integer type). -**ts_created Date Formats** +#### `ts_created` Date Formats The `ts_created` field accepts date strings in the following formats: - `'Feb 18, 2026 @ 22:32:02.960'` (with milliseconds) - `'Feb 18, 2026 @ 22:32:02'` (without milliseconds) - `'2026/02/18'` (date only) -**Note:** All date strings are interpreted as UTC for consistent timestamp conversion across different environments. - -**Examples** - -``` -UPDATE SET b_is_valid=false WHERE s_test_case_name='test1' -UPDATE SET b_is_valid=false WHERE s_gpu_type!='H100' -UPDATE SET b_is_valid=false WHERE d_latency > 100.5 AND l_count >= 10 -UPDATE SET b_is_valid=false WHERE ts_created <= 'Feb 18, 2026 @ 22:32:02.960' AND s_test_case_name='test1' -``` +All date strings are interpreted as UTC for consistent timestamp conversion. diff --git a/jenkins/scripts/perf/disaggregated/slurm_launch_draft.sh b/jenkins/scripts/perf/disaggregated/slurm_launch_draft.sh index 1ff55863eaf5..1930563be027 100644 --- a/jenkins/scripts/perf/disaggregated/slurm_launch_draft.sh +++ b/jenkins/scripts/perf/disaggregated/slurm_launch_draft.sh @@ -21,7 +21,7 @@ echo "Starting gen servers..." for i in $(seq 0 $((numGenServers - 1))); do gen_world_size=$((nodesPerGenServer * gpusPerNodePerGenServer)) export DISAGG_SERVING_TYPE="GEN_$i" - export pytestCommand="$pytestCommandWorker" + export pytestCommand="$pytestCommandGENWorker" srun "${srunArgs[@]}" --kill-on-bad-exit=1 \ -N $nodesPerGenServer \ --ntasks=$gen_world_size \ @@ -36,7 +36,7 @@ if [ "${TRTLLM_DISAGG_BENCHMARK_GEN_ONLY:-0}" != "1" ]; then for i in $(seq 0 $((numCtxServers - 1))); do ctx_world_size=$((nodesPerCtxServer * gpusPerNodePerCtxServer)) export DISAGG_SERVING_TYPE="CTX_$i" - export pytestCommand="$pytestCommandWorker" + export pytestCommand="$pytestCommandCTXWorker" srun "${srunArgs[@]}" --kill-on-bad-exit=1 \ -N $nodesPerCtxServer \ --ntasks=$ctx_world_size \ diff --git a/jenkins/scripts/perf/disaggregated/submit.py b/jenkins/scripts/perf/disaggregated/submit.py index 4233ba173bfd..6962fa2c4e05 100644 --- a/jenkins/scripts/perf/disaggregated/submit.py +++ b/jenkins/scripts/perf/disaggregated/submit.py @@ -4,7 +4,8 @@ import yaml -DISAGG_CONFIG_FOLDER = "tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity" +AGG_CONFIG_FOLDER = "tests/scripts/perf-sanity/aggregated" +DISAGG_CONFIG_FOLDER = "tests/scripts/perf-sanity/disaggregated" def get_hardware_config(config, benchmark_mode): @@ -222,19 +223,34 @@ def is_output_file_part(part): ) -def parse_test_case_name(test_list_path, llm_src): +def parse_test_case_name(test_list_path, llm_src, split_group=0): """Parse test list to get config yaml path and benchmark mode. Test formats for disagg: - Disagg e2e: disagg_upload-e2e-{config_base} - Disagg gen_only: disagg_upload-gen_only-{config_base} + Args: + test_list_path: Path to the test list file. + llm_src: Path to the LLM source code. + split_group: 1-indexed split group id. When > 0, selects the + split_group-th test from the list instead of the first one. + Returns: tuple: (config_yaml_path, benchmark_mode) - benchmark_mode: "e2e" or "gen_only" """ with open(test_list_path, "r") as f: - first_line = f.readline().strip() + lines = [line.strip() for line in f if line.strip()] + + if split_group > 0: + if split_group > len(lines): + raise ValueError( + f"split_group {split_group} exceeds number of tests in test list ({len(lines)})" + ) + first_line = lines[split_group - 1] + else: + first_line = lines[0] if "[" not in first_line or "]" not in first_line: raise ValueError( @@ -303,10 +319,18 @@ def main(): default="", help="Path to file containing srun args (optional, CI mode only)", ) + parser.add_argument( + "--split-group", + type=int, + default=0, + help="1-indexed split group id. Selects the N-th test from the test list.", + ) args = parser.parse_args() - config_yaml, benchmark_mode = parse_test_case_name(args.test_list, args.llm_src) + config_yaml, benchmark_mode = parse_test_case_name( + args.test_list, args.llm_src, args.split_group + ) with open(config_yaml, "r") as f: config = yaml.safe_load(f) @@ -341,30 +365,47 @@ def main(): benchmark_pytest_command, ) = get_pytest_commands(script_prefix_lines) - # Build worker env vars, add extra env vars for gen_only mode - worker_env_vars = env_config["worker_env_var"] + # Build worker env vars (split into ctx and gen for role-specific settings) + base_worker_env_vars = ( + f"FLASHINFER_JIT_DIR=/tmp/flashinfer_jit_cache_\\${{SLURM_LOCALID}} " + f"HF_HOME=/tmp/hf_home " + f"{env_config['worker_env_var']}" + ) + ctx_worker_env_vars = base_worker_env_vars + gen_worker_env_vars = base_worker_env_vars server_env_vars = env_config["server_env_var"] # Handle gen only mode if "gen_only_no_context" in benchmark_mode: - worker_env_vars = f"TRTLLM_DISAGG_BENCHMARK_GEN_ONLY=1 {worker_env_vars}" + gen_worker_env_vars = f"TRTLLM_DISAGG_BENCHMARK_GEN_ONLY=1 {gen_worker_env_vars}" server_env_vars = f"TRTLLM_DISAGG_BENCHMARK_GEN_ONLY=1 {server_env_vars}" script_prefix_lines.append("export TRTLLM_DISAGG_BENCHMARK_GEN_ONLY=1") srun_args_lines.append("--container-env=TRTLLM_DISAGG_BENCHMARK_GEN_ONLY") elif "gen_only" in benchmark_mode: concurrency = benchmark_config.get("concurrency", 1) - worker_env_vars = ( + ctx_worker_env_vars = f"TRTLLM_DISABLE_KV_CACHE_TRANSFER_OVERLAP=1 {ctx_worker_env_vars}" + gen_worker_env_vars = ( f"TRTLLM_DISABLE_KV_CACHE_TRANSFER_OVERLAP=1 " - f"TLLM_BENCHMARK_REQ_QUEUES_SIZE={concurrency} {worker_env_vars}" + f"TLLM_BENCHMARK_REQ_QUEUES_SIZE={concurrency} {gen_worker_env_vars}" ) + pytest_common_vars = "" + script_prefix_lines.extend( [ worker_pytest_command, disagg_server_pytest_command, benchmark_pytest_command, - f'export pytestCommandWorker="unset UCX_TLS && {worker_env_vars} $partialPytestCommandWorker"', - f'export pytestCommandDisaggServer="{server_env_vars} $partialPytestCommandDisaggServer"', - f'export pytestCommandBenchmark="{env_config["benchmark_env_var"]} $partialPytestCommandBenchmark"', + f'export PYTEST_COMMON_VARS="{pytest_common_vars}"', + f'export CTX_WORKER_ENV_VARS="{ctx_worker_env_vars}"', + f'export GEN_WORKER_ENV_VARS="{gen_worker_env_vars}"', + f'export SERVER_ENV_VARS="{server_env_vars}"', + f'export BENCHMARK_ENV_VARS="{env_config["benchmark_env_var"]}"', + 'export pytestCommandCTXWorker="unset UCX_TLS && $CTX_WORKER_ENV_VARS' + ' $PYTEST_COMMON_VARS $partialPytestCommandWorker"', + 'export pytestCommandGENWorker="unset UCX_TLS && $GEN_WORKER_ENV_VARS' + ' $PYTEST_COMMON_VARS $partialPytestCommandWorker"', + 'export pytestCommandDisaggServer="$SERVER_ENV_VARS $PYTEST_COMMON_VARS $partialPytestCommandDisaggServer"', + 'export pytestCommandBenchmark="$BENCHMARK_ENV_VARS $PYTEST_COMMON_VARS $partialPytestCommandBenchmark"', f"export runScript={args.run_sh}", f"export installScript={install_script}", f"export configYamlPath={config_yaml}", diff --git a/jenkins/scripts/perf/get_pre_merge_html.py b/jenkins/scripts/perf/get_pre_merge_html.py new file mode 100644 index 000000000000..927675b54e86 --- /dev/null +++ b/jenkins/scripts/perf/get_pre_merge_html.py @@ -0,0 +1,276 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +"""Generate a pre-merge HTML report with inline SVG performance charts. + +Reads perf_data.yaml files produced by test stages, queries OpenSearch for +historical data and baselines, then generates an HTML report visualizing +key throughput metrics with history, new data, baseline, and threshold lines +for regression comparison. +""" + +import argparse +import os +from html import escape as escape_html + +import yaml + +# Set OPEN_SEARCH_DB_BASE_URL before importing perf_utils, because +# open_search_db captures the env var at module-import time. +if not os.environ.get("OPEN_SEARCH_DB_BASE_URL"): + os.environ["OPEN_SEARCH_DB_BASE_URL"] = "http://gpuwa.nvidia.com" + +from perf_utils import ( + CHART_METRICS, + METRIC_LABELS, + _extract_points, + _generate_svg_chart, + _get_threshold_for_metric, + _ts_to_date, + get_history_data, +) + +# --------------------------------------------------------------------------- +# Data gathering +# --------------------------------------------------------------------------- + + +def load_perf_data(input_files): + """Read comma-separated perf_data.yaml paths and return a flat list of new_data dicts.""" + yaml_files = [f.strip() for f in input_files.split(",") if f.strip()] + all_new_data = [] + load_failures = 0 + for yaml_file in yaml_files: + try: + with open(yaml_file, "r", encoding="utf-8") as f: + content = yaml.safe_load(f) + if content is None or not isinstance(content, list): + continue + for e in content: + if not isinstance(e, dict): + continue + nd = e.get("new_data") + if isinstance(nd, dict) and "s_test_case_name" in nd: + all_new_data.append(nd) + except (OSError, yaml.YAMLError, UnicodeDecodeError) as exc: + load_failures += 1 + print(f"Warning: Failed to load {yaml_file}: {exc}") + if yaml_files and not all_new_data and load_failures == len(yaml_files): + raise RuntimeError("Failed to load any perf data YAML inputs; cannot generate report.") + return all_new_data + + +# --------------------------------------------------------------------------- +# History data query +# --------------------------------------------------------------------------- + + +def get_pre_merge_history_data(new_data_list): + """Query OpenSearch for history data matching test cases in *new_data_list*. + + Uses :func:`perf_utils.get_history_data` to fetch post-merge history + (both baseline and non-baseline), then filters to only the + (s_test_case_name, s_gpu_type) pairs present in *new_data_list*. + + Returns: + dict mapping (test_case, gpu_type) -> { + "history_data": [...], + "baseline_data": [...], + } + or empty dict on failure / no matches. + """ + if not new_data_list: + return {} + + # Determine which test case keys are present in new data + needed_keys = set() + for nd in new_data_list: + key = (nd.get("s_test_case_name", ""), nd.get("s_gpu_type", "")) + needed_keys.add(key) + + grouped = get_history_data( + extra_must_clauses=[ + {"term": {"b_is_post_merge": True}}, + {"term": {"s_branch": "main"}}, + ] + ) + + if grouped is None: + print("Warning: Failed to query history data from OpenSearch") + return {} + + # Filter to only the test cases we have new data for + filtered = {} + for key, bucket in grouped.items(): + if key in needed_keys: + filtered[key] = bucket + + return filtered + + +# --------------------------------------------------------------------------- +# HTML report generation +# --------------------------------------------------------------------------- + + +def _extract_simple_points(data_list, metric): + """Extract (datetime, float_value) pairs from a list of data dicts.""" + points = [] + for d in data_list: + ts = d.get("ts_created") or d.get("@timestamp") + val = d.get(metric) + if ts is not None and val is not None: + try: + points.append((_ts_to_date(ts), float(val))) + except (ValueError, TypeError): + pass + points.sort(key=lambda p: p[0]) + return points + + +def generate_pre_merge_html(new_data_list, history_grouped, output_file): + """Generate HTML report visualizing new data against history + baseline. + + For each (test_case, gpu_type) present in *new_data_list*, renders 4 + charts (one per key metric) showing history line, new data points, + baseline line, and threshold line for regression comparison. + """ + # Group new data by (test_case, gpu_type) + new_groups = {} + for nd in new_data_list: + key = (nd.get("s_test_case_name", ""), nd.get("s_gpu_type", "")) + new_groups.setdefault(key, []).append(nd) + + sections_html = [] + for (test_case, gpu_type), new_data_entries in sorted(new_groups.items()): + bucket = history_grouped.get((test_case, gpu_type), {}) + history_data = bucket.get("history_data", []) + baseline_data_list = bucket.get("baseline_data", []) + + charts = [] + for metric in CHART_METRICS: + label = METRIC_LABELS.get(metric, metric) + + # History points (blue line) — use 3-tuple version from perf_utils + hist_pts = _extract_points(history_data, metric) + + # New data points (red dots) + new_pts = _extract_simple_points(new_data_entries, metric) + + # Baseline value from the latest baseline entry + baseline_value = None + if baseline_data_list: + latest_bl = baseline_data_list[-1] + bl_val = latest_bl.get(metric) + if bl_val is not None: + baseline_value = float(bl_val) + + # Threshold line value + threshold_line_value = None + if baseline_value is not None: + threshold = _get_threshold_for_metric(baseline_data_list, metric) + threshold_line_value = baseline_value * (1 - threshold) + + charts.append( + _generate_svg_chart( + hist_pts, + metric, + label, + new_points=new_pts, + baseline_value=baseline_value, + threshold_line_value=threshold_line_value, + ) + ) + + header = escape_html(f"{test_case} [{gpu_type}]") + section = f""" +
+ {header} +
+ {"".join(charts)} +
+
+ """ + sections_html.append(section) + + total_new = len(new_data_list) + html = f""" + + + + Perf Sanity Pre-Merge Results + + + +

Perf Sanity Pre-Merge Results

+

{len(new_groups)} test case(s) · {total_new} new data point(s)

+ {"".join(sections_html)} + + +""" + with open(output_file, "w", encoding="utf-8") as f: + f.write(html) + + print(f"Generated pre-merge perf report with {len(new_groups)} test cases: {output_file}") + + +# --------------------------------------------------------------------------- +# CLI +# --------------------------------------------------------------------------- + + +def main(): + parser = argparse.ArgumentParser( + description="Generate a pre-merge HTML report with historical " + "performance charts, baseline, and threshold lines." + ) + parser.add_argument( + "--input-files", + type=str, + required=True, + help="Comma-separated list of perf_data.yaml paths", + ) + parser.add_argument("--output-file", type=str, required=True, help="Output HTML file path") + args = parser.parse_args() + + new_data_list = load_perf_data(args.input_files) + history_grouped = get_pre_merge_history_data(new_data_list) + generate_pre_merge_html(new_data_list, history_grouped, args.output_file) + + +if __name__ == "__main__": + main() diff --git a/jenkins/scripts/perf/local/README.md b/jenkins/scripts/perf/local/README.md index ccbdbba833f0..d11e9d7b2297 100644 --- a/jenkins/scripts/perf/local/README.md +++ b/jenkins/scripts/perf/local/README.md @@ -1,8 +1,28 @@ # Local SLURM Launch Scripts -You can use `python3 submit.py ... ` to generate slurm scripts. +## Overview -Then launch the job: `sbatch {timestamp}/slurm_launch.sh`. +This directory contains scripts for running perf sanity tests locally via SLURM. The workflow has three steps: + +1. **`submit.py`** generates a complete `slurm_launch.sh` script. It reads the test config YAML, detects aggregated vs disaggregated mode, and combines SBATCH parameters + environment variables + the appropriate draft template (`jenkins/scripts/perf/aggregated/slurm_launch_draft.sh` or `jenkins/scripts/perf/disaggregated/slurm_launch_draft.sh`) into a single launch script. A `test_list.txt` is also written to the work directory. + +2. **`sbatch slurm_launch.sh`** submits the job to SLURM. Inside the launch script: + - For **aggregated** mode, a single `srun` invokes `slurm_run.sh`. + - For **disaggregated** mode, `srun` first runs `slurm_install.sh` on all nodes, then launches separate `srun` commands for gen workers, ctx workers, the disagg server, and the benchmark client. + +3. **`slurm_install.sh`** handles build and installation inside the container. It optionally builds the TensorRT-LLM wheel (when `--build-wheel` is set) and then runs `pip install -e .` plus dev requirements. A lock-file mechanism ensures only one process per node performs the install while others wait. + +4. **`slurm_run.sh`** runs the pytest command. In aggregated mode, it first sources `slurm_install.sh` to run the install step, then executes the pytest command. In disaggregated mode, the install has already been done by the launch script, so `slurm_run.sh` runs pytest directly. + +``` +submit.py + | + v +slurm_launch.sh (generated) + | + |-- srun --> slurm_install.sh (build wheel + pip install) + |-- srun --> slurm_run.sh (run pytest) +``` ## Optional Arguments @@ -19,122 +39,63 @@ Then launch the job: `sbatch {timestamp}/slurm_launch.sh`. - `--llm-src`: Path to LLM source code. - `--build-wheel`: Add this flag to build the wheel before running tests. - `--install-mode`: Installation mode - `source` (pip install -e ., default) or `wheel` (pip install *.whl). +- `--capture-nsys`: Add this flag to capture an nsys profile during the test run. +- `--nsys-start-stop`: Nsys start-stop range (default: `1-100`). +- `--ctx-nsys-start-stop`: CTX Worker Nsys start-stop range (default: `1-100`). +- `--gen-nsys-start-stop`: GEN Worker Nsys start-stop range (default: `1-100`). `--image` can be obtained by: ```bash +# B200 +image=$(grep LLM_DOCKER_IMAGE $trtllm/jenkins/current_image_tags.properties | head -1 | awk -F "=" '{print $2}' ) +image=$(echo $image | sed 's|urm.nvidia.com/|urm.nvidia.com#|g') +# GB200 image=$(grep LLM_SBSA_DOCKER_IMAGE $trtllm/jenkins/current_image_tags.properties | head -1 | awk -F "=" '{print $2}' ) image=$(echo $image | sed 's|urm.nvidia.com/|urm.nvidia.com#|g') ``` -## OCI - -### Aggregated Mode - -Using `--test-list`: - -```bash -python3 submit.py --test-list "perf/test_perf_sanity.py::test_e2e[aggr-deepseek_r1_fp4_v2_2_nodes_grace_blackwell-r1_fp4_v2_tep8_mtp3]" \ - --partition batch \ - --account coreai_comparch_trtllm \ - --job-name aggr_test \ - --image "urm.nvidia.com#sw-tensorrt-docker/tensorrt-llm:pytorch-25.12-py3-aarch64-ubuntu24.04-trt10.14.1.48-skip-tritondevel-202602011118-10901" \ - --mounts $mounts \ - --llm-models-root $llm_models_path -``` - -Using `--config-file` and `--test-name`: - -```bash -python3 submit.py --config-file $trtllm/tests/scripts/perf-sanity/deepseek_r1_fp4_v2_2_nodes_grace_blackwell.yaml \ - --test-name r1_fp4_v2_tep8_mtp3 \ - --partition batch \ - --account coreai_comparch_trtllm \ - --job-name aggr_test \ - --image "urm.nvidia.com#sw-tensorrt-docker/tensorrt-llm:pytorch-25.12-py3-aarch64-ubuntu24.04-trt10.14.1.48-skip-tritondevel-202602011118-10901" \ - --mounts $mounts \ - --llm-models-root $llm_models_path -``` - -### Disaggregated Mode - -Using `--test-list`: - -```bash -python3 submit.py --test-list "perf/test_perf_sanity.py::test_e2e[disagg-gb200-deepseek-r1-fp4_1k1k_ctx1_dep4_gen1_dep4_eplb0_mtp1_ccb-UCX]" \ - --partition batch \ - --account coreai_comparch_trtllm \ - --job-name disagg_test \ - --image "urm.nvidia.com#sw-tensorrt-docker/tensorrt-llm:pytorch-25.12-py3-aarch64-ubuntu24.04-trt10.14.1.48-skip-tritondevel-202602011118-10901" \ - --mounts $mounts \ - --llm-models-root $llm_models_path -``` - -Using `--config-file`: +## Cluster Settings -```bash -python3 submit.py --config-file $trtllm/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200-deepseek-r1-fp4_1k1k_ctx1_dep4_gen1_dep4_eplb0_mtp1_ccb-UCX.yaml \ - --benchmark-mode gen_only \ - --partition batch \ - --account coreai_comparch_trtllm \ - --job-name disagg_test \ - --image "urm.nvidia.com#sw-tensorrt-docker/tensorrt-llm:pytorch-25.12-py3-aarch64-ubuntu24.04-trt10.14.1.48-skip-tritondevel-202602011118-10901" \ - --mounts $mounts \ - --llm-models-root $llm_models_path -``` +| Cluster | `--partition` | `--account` | +|---------|---------------|-------------| +| OCI | `batch` | `coreai_comparch_trtllm` | +| DLCluster | `gb200nvl72_preprod` | `coreai_comparch_trtllm` | -## DLCluster +## Examples ### Aggregated Mode -Using `--test-list`: - ```bash python3 submit.py --test-list "perf/test_perf_sanity.py::test_e2e[aggr-deepseek_r1_fp4_v2_2_nodes_grace_blackwell-r1_fp4_v2_tep8_mtp3]" \ - --partition gb200nvl72_preprod \ - --account coreai_comparch_trtllm \ - --job-name coreai_comparch_trtllm \ - --image "urm.nvidia.com#sw-tensorrt-docker/tensorrt-llm:pytorch-25.12-py3-aarch64-ubuntu24.04-trt10.14.1.48-skip-tritondevel-202602011118-10901" \ - --mounts $mounts \ - --llm-models-root $llm_models_path -``` - -Using `--config-file` and `--test-name`: - -```bash -python3 submit.py --config-file $trtllm/tests/scripts/perf-sanity/deepseek_r1_fp4_v2_2_nodes_grace_blackwell.yaml \ - --test-name r1_fp4_v2_tep8_mtp3 \ - --partition gb200nvl72_preprod \ - --account coreai_comparch_trtllm \ - --job-name coreai_comparch_trtllm \ - --image "urm.nvidia.com#sw-tensorrt-docker/tensorrt-llm:pytorch-25.12-py3-aarch64-ubuntu24.04-trt10.14.1.48-skip-tritondevel-202602011118-10901" \ + --draft-launch-sh $trtllm/jenkins/scripts/perf/aggregated/slurm_launch_draft.sh \ + --launch-sh $work_dir/slurm_launch.sh \ + --install-sh $trtllm/jenkins/scripts/perf/local/slurm_install.sh \ + --run-sh $trtllm/jenkins/scripts/perf/local/slurm_run.sh \ + --llm-src $trtllm \ + --work-dir $work_dir \ + --partition $partition \ + --account $account \ + --job-name aggr_test \ + --image $image \ --mounts $mounts \ --llm-models-root $llm_models_path ``` ### Disaggregated Mode -Using `--test-list`: - ```bash -python3 submit.py --test-list "perf/test_perf_sanity.py::test_e2e[disagg-gb200-deepseek-r1-fp4_1k1k_ctx1_dep4_gen1_dep4_eplb0_mtp1_ccb-UCX]" \ - --partition gb200nvl72_preprod \ - --account coreai_comparch_trtllm \ - --job-name coreai_comparch_trtllm \ - --image "urm.nvidia.com#sw-tensorrt-docker/tensorrt-llm:pytorch-25.12-py3-aarch64-ubuntu24.04-trt10.14.1.48-skip-tritondevel-202602011118-10901" \ - --mounts $mounts \ - --llm-models-root $llm_models_path -``` - -Using `--config-file`: - -```bash -python3 submit.py --config-file $trtllm/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200-deepseek-r1-fp4_1k1k_ctx1_dep4_gen1_dep4_eplb0_mtp1_ccb-UCX.yaml \ - --benchmark-mode gen_only \ - --partition gb200nvl72_preprod \ - --account coreai_comparch_trtllm \ - --job-name coreai_comparch_trtllm \ - --image "urm.nvidia.com#sw-tensorrt-docker/tensorrt-llm:pytorch-25.12-py3-aarch64-ubuntu24.04-trt10.14.1.48-skip-tritondevel-202602011118-10901" \ +python3 submit.py --test-list "perf/test_perf_sanity.py::test_e2e[disagg-e2e-gb200_deepseek-r1-fp4_1k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX]" \ + --draft-launch-sh $trtllm/jenkins/scripts/perf/disaggregated/slurm_launch_draft.sh \ + --launch-sh $work_dir/slurm_launch.sh \ + --install-sh $trtllm/jenkins/scripts/perf/local/slurm_install.sh \ + --run-sh $trtllm/jenkins/scripts/perf/local/slurm_run.sh \ + --llm-src $trtllm \ + --work-dir $work_dir \ + --partition $partition \ + --account $account \ + --job-name disagg_test \ + --image $image \ --mounts $mounts \ --llm-models-root $llm_models_path ``` diff --git a/jenkins/scripts/perf/local/submit.py b/jenkins/scripts/perf/local/submit.py index a13fba1218bb..0afcb17806d3 100755 --- a/jenkins/scripts/perf/local/submit.py +++ b/jenkins/scripts/perf/local/submit.py @@ -6,8 +6,10 @@ import yaml -AGGR_CONFIG_FOLDER = "tests/scripts/perf-sanity" -DISAGG_CONFIG_FOLDER = "tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity" +AGG_CONFIG_FOLDER = os.environ.get("AGG_CONFIG_FOLDER", "tests/scripts/perf-sanity/aggregated") +DISAGG_CONFIG_FOLDER = os.environ.get( + "DISAGG_CONFIG_FOLDER", "tests/scripts/perf-sanity/disaggregated" +) def get_llm_src_default(): @@ -111,9 +113,12 @@ def get_config_yaml_path(llm_src, config_base_name, benchmark_mode): str: Full path to config yaml file """ if benchmark_mode in ("e2e", "gen_only", "ctx_only"): - config_dir = os.path.join(llm_src, DISAGG_CONFIG_FOLDER) + config_dir = DISAGG_CONFIG_FOLDER else: - config_dir = os.path.join(llm_src, AGGR_CONFIG_FOLDER) + config_dir = AGG_CONFIG_FOLDER + # If relative path, join with llm root + if not os.path.isabs(config_dir): + config_dir = os.path.join(llm_src, config_dir) config_yaml_path = os.path.join(config_dir, f"{config_base_name}.yaml") @@ -402,6 +407,22 @@ def main(): choices=["source", "wheel"], help="Installation mode: source (pip install -e ., default) or wheel (pip install *.whl)", ) + parser.add_argument("--capture-nsys", action="store_true", help="Capture nsys profile") + parser.add_argument( + "--nsys-start-stop", + default="1-100", + help="Nsys start-stop range for aggregated mode (default: 1-100)", + ) + parser.add_argument( + "--ctx-nsys-start-stop", + default="1-100", + help="Nsys start-stop range for context workers in disaggregated mode (default: 1-100)", + ) + parser.add_argument( + "--gen-nsys-start-stop", + default="1-100", + help="Nsys start-stop range for generation workers in disaggregated mode (default: 1-100)", + ) args = parser.parse_args() @@ -523,40 +544,108 @@ def main(): ] ) + nsys_prefix = "" + tllm_profile_start_stop = "" + ctx_tllm_profile_start_stop = "" + gen_tllm_profile_start_stop = "" + if args.capture_nsys: + if runtime_mode == "disaggregated": + nsys_output = f"{work_dir}/nsys.%q{{DISAGG_SERVING_TYPE}}.rank%q{{SLURM_PROCID}}" + else: + nsys_output = f"{work_dir}/nsys.rank%q{{SLURM_PROCID}}" + nsys_prefix = ( + "nsys profile" + " -t cuda,nvtx,python-gil" + " --sample cpu" + " --cuda-graph-trace node" + " -e TLLM_PROFILE_RECORD_GC=1,TLLM_LLMAPI_ENABLE_NVTX=1,TLLM_TORCH_PROFILE_TRACE=trace.json" + " --trace-fork-before-exec=true" + " -f true" + " --gpu-metrics-devices=none" + " -c cudaProfilerApi" + " --capture-range-end=stop" + " --export=sqlite" + f" -o {nsys_output}" + ) + tllm_profile_start_stop = args.nsys_start_stop + ctx_tllm_profile_start_stop = args.ctx_nsys_start_stop + gen_tllm_profile_start_stop = args.gen_nsys_start_stop + pytest_common_vars = ( f"LLM_ROOT='{llm_src}' " f"LLM_BACKEND_ROOT='{llm_src}/triton_backend' " f"LLM_MODELS_ROOT='{args.llm_models_root}' " + f"AGG_CONFIG_FOLDER='{AGG_CONFIG_FOLDER}' " + f"DISAGG_CONFIG_FOLDER='{DISAGG_CONFIG_FOLDER}' " ) llmapi_launch = f"{llm_src}/tensorrt_llm/llmapi/trtllm-llmapi-launch" + # Add shared exports + script_prefix_lines.extend( + [ + f"export CAPTURE_NSYS={'true' if args.capture_nsys else 'false'}", + f'export NSYS_PREFIX="{nsys_prefix}"', + f'export LLM_API_LAUNCH="{llmapi_launch}"', + f'export PYTEST_COMMON_VARS="{pytest_common_vars}"', + f'export PYTEST_COMMAND="{pytest_command}"', + ] + ) + + server_env_vars = "" + benchmark_env_var = "" if runtime_mode == "disaggregated": - # Build worker env vars - worker_env_vars = env_config.get("worker_env_var", "") + # Build worker env vars (split into ctx and gen for role-specific settings) + common_worker_env_var = env_config.get("worker_env_var", "") + ctx_worker_env_vars = ( + f"TLLM_PROFILE_START_STOP='{ctx_tllm_profile_start_stop}' " + f"FLASHINFER_JIT_DIR=/tmp/flashinfer_jit_cache_\\${{SLURM_LOCALID}} " + f"HF_HOME=/tmp/hf_home " + f"{common_worker_env_var}" + ) + gen_worker_env_vars = ( + f"TLLM_PROFILE_START_STOP='{gen_tllm_profile_start_stop}' " + f"FLASHINFER_JIT_DIR=/tmp/flashinfer_jit_cache_\\${{SLURM_LOCALID}} " + f"HF_HOME=/tmp/hf_home " + f"{common_worker_env_var}" + ) server_env_vars = env_config.get("server_env_var", "") benchmark_env_var = env_config.get("benchmark_env_var", "") # Handle gen only mode if "gen_only_no_context" in bm_config.get("mode", ""): - worker_env_vars = f"TRTLLM_DISAGG_BENCHMARK_GEN_ONLY=1 {worker_env_vars}" + gen_worker_env_vars = f"TRTLLM_DISAGG_BENCHMARK_GEN_ONLY=1 {gen_worker_env_vars}" server_env_vars = f"TRTLLM_DISAGG_BENCHMARK_GEN_ONLY=1 {server_env_vars}" script_prefix_lines.append("export TRTLLM_DISAGG_BENCHMARK_GEN_ONLY=1") srun_args_lines.append("--container-env=TRTLLM_DISAGG_BENCHMARK_GEN_ONLY") elif "gen_only" in bm_config.get("mode", ""): concurrency = bm_config.get("concurrency", 1) - worker_env_vars = ( + ctx_worker_env_vars = ( + f"TRTLLM_DISABLE_KV_CACHE_TRANSFER_OVERLAP=1 {ctx_worker_env_vars}" + ) + gen_worker_env_vars = ( f"TRTLLM_DISABLE_KV_CACHE_TRANSFER_OVERLAP=1 " - f"TLLM_BENCHMARK_REQ_QUEUES_SIZE={concurrency} {worker_env_vars}" + f"TLLM_BENCHMARK_REQ_QUEUES_SIZE={concurrency} {gen_worker_env_vars}" ) - pytest_cmd_worker = ( - f"unset UCX_TLS && {worker_env_vars} {pytest_common_vars} " - f"{llmapi_launch} {pytest_command} --junitxml={work_dir}/report.xml" - ) script_prefix_lines.extend( [ - f'export pytestCommandWorker="{pytest_cmd_worker}"', - f'export pytestCommandDisaggServer="{server_env_vars} {pytest_common_vars} {pytest_command}"', - f'export pytestCommandBenchmark="{benchmark_env_var} {pytest_common_vars} {pytest_command}"', + f'export CTX_WORKER_ENV_VARS="{ctx_worker_env_vars}"', + f'export GEN_WORKER_ENV_VARS="{gen_worker_env_vars}"', + f'export SERVER_ENV_VARS="{server_env_vars}"', + f'export BENCHMARK_ENV_VARS="{benchmark_env_var}"', + ( + 'export pytestCommandCTXWorker="unset UCX_TLS &&' + " $CTX_WORKER_ENV_VARS $PYTEST_COMMON_VARS" + " $NSYS_PREFIX $LLM_API_LAUNCH" + f' $PYTEST_COMMAND --junitxml={work_dir}/report.xml"' + ), + ( + 'export pytestCommandGENWorker="unset UCX_TLS &&' + " $GEN_WORKER_ENV_VARS $PYTEST_COMMON_VARS" + " $NSYS_PREFIX $LLM_API_LAUNCH" + f' $PYTEST_COMMAND --junitxml={work_dir}/report.xml"' + ), + 'export pytestCommandDisaggServer="$SERVER_ENV_VARS $PYTEST_COMMON_VARS $PYTEST_COMMAND"', + 'export pytestCommandBenchmark="$BENCHMARK_ENV_VARS $PYTEST_COMMON_VARS $PYTEST_COMMAND"', f"export numCtxServers={hardware_config.get('num_ctx_servers', '')}", f"export numGenServers={hardware_config.get('num_gen_servers', '')}", f"export gpusPerNode={hardware_config.get('gpus_per_node', '')}", @@ -579,12 +668,18 @@ def main(): ] ) else: + worker_env_vars = ( + f"TLLM_PROFILE_START_STOP='{tllm_profile_start_stop}' " + f"FLASHINFER_JIT_DIR=/tmp/flashinfer_jit_cache_\\${{SLURM_LOCALID}} " + f"HF_HOME=/tmp/hf_home " + ) # Aggregated mode (including ctx_only) script_prefix_lines.extend( [ + f'export WORKER_ENV_VARS="{worker_env_vars}"', ( - f'export pytestCommand="{pytest_common_vars} {llmapi_launch} ' - f'{pytest_command} --junitxml={work_dir}/report.xml"' + 'export pytestCommand="$WORKER_ENV_VARS $PYTEST_COMMON_VARS $NSYS_PREFIX $LLM_API_LAUNCH' + f' $PYTEST_COMMAND --junitxml={work_dir}/report.xml"' ), f"export gpusPerNode={hardware_config.get('gpus_per_node', '')}", f"export gpusPerNodePerServer={hardware_config.get('gpus_per_node_per_server', '')}", diff --git a/jenkins/scripts/perf/perf_regression.py b/jenkins/scripts/perf/perf_regression.py deleted file mode 100644 index 0f4a48db435e..000000000000 --- a/jenkins/scripts/perf/perf_regression.py +++ /dev/null @@ -1,275 +0,0 @@ -#!/usr/bin/env python3 -"""Merge perf regression info from multiple YAML files into an HTML report.""" - -import argparse -from html import escape as escape_html - -import yaml - -# Metrics where larger is better -MAXIMIZE_METRICS = [ - "d_seq_throughput", - "d_token_throughput", - "d_total_token_throughput", - "d_user_throughput", - "d_mean_tpot", - "d_median_tpot", - "d_p99_tpot", -] - -# Metrics where smaller is better -MINIMIZE_METRICS = [ - "d_mean_ttft", - "d_median_ttft", - "d_p99_ttft", - "d_mean_itl", - "d_median_itl", - "d_p99_itl", - "d_mean_e2el", - "d_median_e2el", - "d_p99_e2el", -] - - -def _get_metric_keys(): - """Get all metric-related keys for filtering config keys.""" - metric_keys = set() - for metric in MAXIMIZE_METRICS + MINIMIZE_METRICS: - metric_suffix = metric[2:] # Strip "d_" prefix - metric_keys.add(metric) - metric_keys.add(f"d_baseline_{metric_suffix}") - metric_keys.add(f"d_threshold_post_merge_{metric_suffix}") - metric_keys.add(f"d_threshold_pre_merge_{metric_suffix}") - return metric_keys - - -def _get_regression_content(data): - """Get regression info and config content as a list of lines.""" - lines = [] - if "s_regression_info" in data: - lines.append("=== Regression Info ===") - regression_info = data["s_regression_info"] - for line in regression_info.split(","): - lines.append(line) - - metric_keys = _get_metric_keys() - - lines.append("") - lines.append("=== Config ===") - config_keys = sorted([key for key in data.keys() if key not in metric_keys]) - for key in config_keys: - if key == "s_regression_info": - continue - value = data[key] - lines.append(f'"{key}": {value}') - - return lines - - -def merge_regression_data(input_files): - """Read all yaml file paths and merge regression data.""" - yaml_files = [f.strip() for f in input_files.split(",") if f.strip()] - - regression_dict = {} - load_failures = 0 - - for yaml_file in yaml_files: - try: - # Path format: .../{stage_name}/{folder_name}/regression_data.yaml - path_parts = yaml_file.replace("\\", "/").split("/") - if len(path_parts) < 3: - continue - - stage_name = path_parts[-3] - folder_name = path_parts[-2] - - with open(yaml_file, "r", encoding="utf-8") as f: - content = yaml.safe_load(f) - if content is None or not isinstance(content, list): - continue - - filtered_data = [ - d for d in content if isinstance(d, dict) and "s_test_case_name" in d - ] - - if not filtered_data: - continue - - if stage_name not in regression_dict: - regression_dict[stage_name] = {} - - if folder_name not in regression_dict[stage_name]: - regression_dict[stage_name][folder_name] = [] - - regression_dict[stage_name][folder_name].extend(filtered_data) - - except (OSError, yaml.YAMLError, UnicodeDecodeError) as e: - load_failures += 1 - print(f"Warning: Failed to load {yaml_file}: {e}") - continue - - # Fail fast if caller provided inputs but none were readable/parseable. - # (Keeps "no regressions found" working when yaml_files is empty.) - if yaml_files and not regression_dict and load_failures == len(yaml_files): - raise RuntimeError("Failed to load any regression YAML inputs; cannot generate report.") - - return regression_dict - - -def generate_html(regression_dict, output_file): - """Generate HTML report from regression data.""" - html_template = """ - - - - Perf Regression Summary - - - -

Perf Regression Summary

- {test_suites} - - - """ - - all_suites_html = [] - total_tests = 0 - - for stage_name in regression_dict: - folder_dict = regression_dict[stage_name] - # Count total tests for this stage - tests_count = sum(len(data_list) for data_list in folder_dict.values()) - total_tests += tests_count - - # Generate summary for the suite - summary = f""" -
-

Stage: {escape_html(stage_name)}

-

Regression Tests: {tests_count}

-
- """ - - # Generate test case details for the suite - test_cases_html = [] - - for folder_name, data_list in folder_dict.items(): - for data in data_list: - test_case_name = data.get("s_test_case_name", "N/A") - test_name = f"perf/test_perf_sanity.py::test_e2e[{folder_name}] - {test_case_name}" - - # Get content lines - content_lines = _get_regression_content(data) - content_html = "".join( - f"{escape_html(line)}" for line in content_lines - ) - - details = f""" -
- {escape_html(test_name)} -
{content_html}
-
- """ - - test_case_html = f""" -
- {details} -
- """ - test_cases_html.append(test_case_html) - - # Combine summary and test cases for this suite - suite_html = f""" -
- {summary} -
- {" ".join(test_cases_html)} -
-
- """ - all_suites_html.append(suite_html) - - # Generate complete HTML - html_content = html_template.format(test_suites="\n".join(all_suites_html)) - - # Write to file - with open(output_file, "w", encoding="utf-8") as f: - f.write(html_content) - - print(f"Generated HTML report with {total_tests} regression entries: {output_file}") - - -def main(): - parser = argparse.ArgumentParser( - description="Merge perf regression info from YAML files into an HTML report." - ) - parser.add_argument( - "--input-files", type=str, required=True, help="Comma-separated list of YAML file paths" - ) - parser.add_argument("--output-file", type=str, required=True, help="Output HTML file path") - args = parser.parse_args() - - regression_dict = merge_regression_data(args.input_files) - generate_html(regression_dict, args.output_file) - - -if __name__ == "__main__": - main() diff --git a/jenkins/scripts/perf/perf_utils.py b/jenkins/scripts/perf/perf_utils.py new file mode 100644 index 000000000000..aae00a35d86a --- /dev/null +++ b/jenkins/scripts/perf/perf_utils.py @@ -0,0 +1,1620 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +"""Shared utilities for perf sanity scripts. + +Contains constants, regression detection algorithms, OpenSearch query helpers, +and HTML/SVG report generation functions used by test.py, get_pre_merge_html.py, +and perf_sanity_triage.py. +""" + +import json as _json +import math +import os +import sys +import time +from collections import defaultdict +from datetime import datetime +from html import escape as escape_html + +sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..")) +from open_search_db import OpenSearchDB + +# --------------------------------------------------------------------------- +# Constants +# --------------------------------------------------------------------------- + +PERF_SANITY_PROJECT_NAME = "swdl-trtllm-infra-ci-prod-perf_sanity_info" +QUERY_LOOKBACK_DAYS = 90 +MAX_QUERY_SIZE = 9999 +DEFAULT_THRESHOLD = 0.05 + +CHART_METRICS = [ + "d_seq_throughput", + "d_token_throughput", + "d_total_token_throughput", + "d_user_throughput", +] + +# Only these 2 metrics determine the overall test-case classification. +CLASSIFICATION_METRICS = [ + "d_token_throughput", + "d_total_token_throughput", +] + +METRIC_LABELS = { + "d_seq_throughput": "Request Throughput (req/s)", + "d_token_throughput": "Output Token Throughput (tok/s)", + "d_total_token_throughput": "Total Token Throughput (tok/s)", + "d_user_throughput": "User Throughput (tok/s)", +} + +# Algorithm parameters +_STABILITY_CV_THRESHOLD = 0.03 # 3% +_REGRESSION_THRESHOLD = 0.05 # 5% +_ROLLING_WINDOW = 7 +_MIN_STABLE_SEGMENT = 7 +_MIN_CONFIRMATION_DAYS = 3 +_DIRECTION_CHANGE_THRESHOLD = 6 # per 30 days +_OUTLIER_ZSCORE = 2.0 + +# Curve type display +_CURVE_TYPE_COLORS = { + "no_regression": "#0d904f", + "sudden_drop": "#d93025", + "gradual_decline": "#e8710a", + "significant_fluctuation": "#7b1fa2", + "occasional_spike": "#c5a600", + "other_reasons": "#607d8b", +} + +_CURVE_TYPE_LABELS = { + "no_regression": "No Regression", + "sudden_drop": "Sudden Drop", + "gradual_decline": "Gradual Decline", + "significant_fluctuation": "Significant Fluctuation", + "occasional_spike": "Occasional Spike", + "other_reasons": "Other Reasons", +} + +# --------------------------------------------------------------------------- +# Timestamp / data utilities +# --------------------------------------------------------------------------- + +_TIME_FORMATS = [ + "%Y-%m-%dT%H:%M:%S.%fZ", + "%Y-%m-%dT%H:%M:%SZ", + "%Y-%m-%dT%H:%M:%S.%f", + "%Y-%m-%dT%H:%M:%S", + "%b %d, %Y @ %H:%M:%S.%f", +] + + +def _parse_timestamp(timestamp): + """Parse a timestamp value into a datetime object.""" + if isinstance(timestamp, (int, float)): + if timestamp > 1e12: + timestamp = timestamp / 1000 + return datetime.fromtimestamp(timestamp) + if isinstance(timestamp, datetime): + return timestamp + timestamp_str = str(timestamp) + for fmt in _TIME_FORMATS: + try: + return datetime.strptime(timestamp_str, fmt) + except ValueError: + continue + return datetime.fromtimestamp(0) + + +def _ts_to_date(ts): + """Convert a millisecond timestamp to a datetime.""" + try: + return datetime.fromtimestamp(int(ts) / 1000) + except (ValueError, TypeError, OSError): + return datetime.fromtimestamp(0) + + +def _extract_points(data_list, metric): + """Extract (datetime, float_value, data_dict) triples from data dicts.""" + points = [] + for d in data_list: + ts = d.get("ts_created") or d.get("@timestamp") + val = d.get(metric) + if ts is not None and val is not None: + try: + points.append((_ts_to_date(ts), float(val), d)) + except (ValueError, TypeError): + pass + points.sort(key=lambda p: p[0]) + return points + + +def _data_dict_to_json_attr(data_dict): + """Serialize a data dict to an HTML-safe JSON string for embedding in attributes.""" + return escape_html(_json.dumps(data_dict, default=str, ensure_ascii=True)) + + +# --------------------------------------------------------------------------- +# Baseline computation +# --------------------------------------------------------------------------- + + +def _daily_aggregate(points): + """Aggregate multiple data points on the same day to a single mean value. + + Args: + points: list of (datetime, float) or (datetime, float, data_dict) + tuples. + + Returns: + list of (date_str, float, [data_dicts]) triples sorted by date. + The third element is a list of original data dicts for that day + (empty list when input items have no third element). + """ + by_day = defaultdict(list) + entries = defaultdict(list) + for item in points: + dt, val = item[0], item[1] + day_key = dt.strftime("%Y-%m-%d") + by_day[day_key].append(val) + if len(item) > 2 and item[2] is not None: + entries[day_key].append(item[2]) + result = [] + for day in sorted(by_day): + vals = by_day[day] + result.append((day, sum(vals) / len(vals), entries[day])) + return result + + +def _rolling_smooth(values, window=3): + """Trailing rolling mean with same-length output. + + Early elements use fewer samples (i.e. the first element is itself, + the second is the mean of the first two, etc.). + """ + if not values: + return [] + smoothed = [] + for i in range(len(values)): + start = max(0, i - window + 1) + w = values[start : i + 1] + smoothed.append(sum(w) / len(w)) + return smoothed + + +def _percentile(values, p): + """Compute the p-th percentile with linear interpolation. + + Args: + values: non-empty list of floats. + p: percentile in [0, 100]. + """ + if not values: + return 0.0 + s = sorted(values) + k = (p / 100.0) * (len(s) - 1) + lo = int(k) + hi = min(lo + 1, len(s) - 1) + frac = k - lo + return s[lo] + frac * (s[hi] - s[lo]) + + +def get_baseline(grouped_data): + """Compute rolling-smooth + P95 baselines and daily data for all entries. + + For each (test_case, gpu_type) key and each metric, this function: + 1. Extracts data points as 3-tuples (datetime, float, data_dict). + 2. Aggregates to daily values preserving original data entries. + 3. Applies rolling smooth (window=3) to daily values. + 4. Computes P95 of the smoothed values as the baseline. + + Mutates ``grouped_data[key]`` to add: + "daily_data": {metric: {"dates": [...], "values": [...], + "entries": [[data_dicts], ...]}}, + "baselines": {metric: float}, + """ + for key, bucket in grouped_data.items(): + history_data = bucket["history_data"] + daily_data = {} + baselines = {} + for metric in CHART_METRICS: + points = _extract_points(history_data, metric) + daily = _daily_aggregate(points) + daily_dates = [d for d, _, _ in daily] + daily_vals = [v for _, v, _ in daily] + daily_entries = [e for _, _, e in daily] + + smoothed = _rolling_smooth(daily_vals, window=3) + baseline = _percentile(smoothed, 95) if smoothed else 0.0 + + daily_data[metric] = { + "dates": daily_dates, + "values": daily_vals, + "entries": daily_entries, + } + baselines[metric] = baseline + bucket["daily_data"] = daily_data + bucket["baselines"] = baselines + + +# --------------------------------------------------------------------------- +# Regression classification +# --------------------------------------------------------------------------- + + +def _extract_jump_commits(daily_entries, daily_dates, js_idx, je_idx): + """Extract commit and timestamp info at jump interval endpoints. + + Args: + daily_entries: list of lists of data_dicts (one list per day). + daily_dates: list of date strings corresponding to daily_entries. + js_idx: jump-start day index (left endpoint). + je_idx: jump-end day index (right endpoint). + + Returns: + {"left": {"s_commit": str, "timestamp": str}, + "right": {"s_commit": str, "timestamp": str}} + or None if data is unavailable. + """ + if not daily_entries or not daily_dates: + return None + js_idx = max(0, min(js_idx, len(daily_entries) - 1)) + je_idx = max(0, min(je_idx, len(daily_entries) - 1)) + + def _pick_last(entries_list): + """Pick the last chronological entry from a day's entries.""" + if not entries_list: + return None + best = entries_list[-1] + for e in entries_list: + ts_e = e.get("ts_created") or e.get("@timestamp", 0) + ts_b = best.get("ts_created") or best.get("@timestamp", 0) + if ts_e is not None and ts_b is not None and ts_e > ts_b: + best = e + commit = best.get("s_commit", "") + ts_raw = best.get("ts_created") or best.get("@timestamp", "") + if isinstance(ts_raw, (int, float)): + if ts_raw > 1e12: + ts_raw = ts_raw / 1000 + ts_str = datetime.fromtimestamp(ts_raw).strftime("%Y-%m-%d %H:%M") + else: + ts_str = str(ts_raw) + return {"s_commit": str(commit), "timestamp": ts_str} + + left = _pick_last(daily_entries[js_idx]) + right = _pick_last(daily_entries[je_idx]) + if left is None and right is None: + return None + return {"left": left, "right": right} + + +def _cv(values): + """Coefficient of variation (std / mean). Returns 0 if mean is 0.""" + if len(values) < 2: + return 0.0 + mean = sum(values) / len(values) + if mean == 0: + return 0.0 + variance = sum((v - mean) ** 2 for v in values) / len(values) + return math.sqrt(variance) / abs(mean) + + +def _is_stable(values, threshold=_STABILITY_CV_THRESHOLD): + """Check if CV < threshold.""" + return _cv(values) < threshold + + +def _rolling_stats(values, window=_ROLLING_WINDOW): + """Compute rolling means, rolling CVs, and direction change count. + + Returns: + (rolling_means, rolling_cvs, direction_changes) + """ + if len(values) < window: + return [], [], 0 + + rolling_means = [] + rolling_cvs = [] + for i in range(len(values) - window + 1): + w = values[i : i + window] + m = sum(w) / len(w) + rolling_means.append(m) + rolling_cvs.append(_cv(w)) + + direction_changes = 0 + for i in range(2, len(rolling_means)): + d_prev = rolling_means[i - 1] - rolling_means[i - 2] + d_curr = rolling_means[i] - rolling_means[i - 1] + if d_prev * d_curr < 0: + direction_changes += 1 + + return rolling_means, rolling_cvs, direction_changes + + +def _find_change_point(values, window=_ROLLING_WINDOW): + """Find the optimal split point using segmented approach (Phase 4). + + Returns: + (split_index, jump_start_index, jump_end_index) or None. + """ + n = len(values) + if n < 2 * window: + return None + + best_score = -1 + best_idx = -1 + eps = 1e-12 + + for i in range(window, n - window + 1): + left = values[:i] + right = values[i:] + left_mean = sum(left) / len(left) + right_mean = sum(right) / len(right) + left_var = sum((v - left_mean) ** 2 for v in left) / len(left) + right_var = sum((v - right_mean) ** 2 for v in right) / len(right) + score = (left_mean - right_mean) ** 2 / (left_var + right_var + eps) + if score > best_score: + best_score = score + best_idx = i + + if best_idx < 0: + return None + + pre_level = sum(values[:best_idx]) / best_idx + post_level = sum(values[best_idx:]) / (n - best_idx) + + if pre_level == post_level: + return best_idx, best_idx, best_idx + + threshold_start = pre_level - 0.2 * (pre_level - post_level) + threshold_end = pre_level - 0.8 * (pre_level - post_level) + + jump_start = best_idx + jump_end = best_idx + + if pre_level > post_level: + for j in range(n): + if values[j] < threshold_start: + jump_start = j + break + for j in range(n): + if values[j] < threshold_end: + jump_end = j + break + else: + for j in range(n): + if values[j] > threshold_start: + jump_start = j + break + for j in range(n): + if values[j] > threshold_end: + jump_end = j + break + + return best_idx, jump_start, jump_end + + +def _is_regression(daily_values, baseline, threshold=_REGRESSION_THRESHOLD): + """Step 1: Determine whether the metric shows a regression. + + A regression exists when the recent average drops more than + ``threshold`` compared to the baseline. + + Returns True if regression is detected, False otherwise. + """ + if not daily_values or baseline == 0: + return False + recent_count = min(5, max(3, len(daily_values))) + recent_avg = sum(daily_values[-recent_count:]) / recent_count + drop_ratio = (baseline - recent_avg) / baseline + return drop_ratio > threshold + + +def _classify_regression_type(daily_values): + """Step 2: Given that a regression exists, determine its subtype. + + Checks in priority order: + 1. Significant Fluctuation + 2. Occasional Spike + 3. Sudden Drop + 4. Gradual Decline + + If none of the four patterns match, falls back to ``"other_reasons"``. + + Returns (regression_type, jump_interval) where regression_type is one of + ``"significant_fluctuation"``, ``"occasional_spike"``, + ``"sudden_drop"``, ``"gradual_decline"``, ``"other_reasons"``. + """ + n_days = len(daily_values) + rolling_means, rolling_cvs, direction_changes = _rolling_stats(daily_values) + + # --- Significant Fluctuation --- + normalized_dir_changes = direction_changes * 30 / n_days if n_days > 0 else 0 + oscillation_windows = 0 + if rolling_means: + for i in range(len(rolling_means)): + w = daily_values[i : i + _ROLLING_WINDOW] + if w and max(w) > 0: + amp = (max(w) - min(w)) / max(w) + if amp > _REGRESSION_THRESHOLD: + oscillation_windows += 1 + has_long_stable = False + stable_run = 0 + for cv_val in rolling_cvs: + if cv_val < _STABILITY_CV_THRESHOLD: + stable_run += 1 + if stable_run >= 2 * _ROLLING_WINDOW: + has_long_stable = True + break + else: + stable_run = 0 + + if ( + normalized_dir_changes > _DIRECTION_CHANGE_THRESHOLD + and oscillation_windows > len(rolling_means) * 0.3 + and not has_long_stable + ): + return "significant_fluctuation", None + + # --- Occasional Spike --- + if n_days >= 3: + mean_val = sum(daily_values) / n_days + std_val = math.sqrt(sum((v - mean_val) ** 2 for v in daily_values) / n_days) + if std_val > 0: + outlier_indices = [ + i + for i, v in enumerate(daily_values) + if abs(v - mean_val) / std_val > _OUTLIER_ZSCORE + ] + else: + outlier_indices = [] + non_outlier_vals = [v for i, v in enumerate(daily_values) if i not in outlier_indices] + if len(outlier_indices) < 3 and non_outlier_vals and _is_stable(non_outlier_vals): + max_consecutive_low = 0 + consecutive = 0 + low_threshold = mean_val - _REGRESSION_THRESHOLD * mean_val + for v in daily_values: + if v < low_threshold: + consecutive += 1 + max_consecutive_low = max(max_consecutive_low, consecutive) + else: + consecutive = 0 + if max_consecutive_low < _MIN_CONFIRMATION_DAYS: + return "occasional_spike", None + + # --- Sudden Drop / Gradual Decline (via change-point analysis) --- + cp = _find_change_point(daily_values) + if cp is not None: + split_idx, jump_start, jump_end = cp + pre_segment = daily_values[:split_idx] + post_segment = daily_values[split_idx:] + + adj_left = max(0, jump_start - 1) + adj_right = jump_end + if adj_left >= adj_right: + adj_left = max(0, adj_right - 1) + if adj_left == adj_right: + adj_right = min(n_days - 1, adj_right + 1) + + if len(pre_segment) >= _MIN_STABLE_SEGMENT and len(post_segment) >= _MIN_CONFIRMATION_DAYS: + pre_stable = _is_stable(pre_segment) + post_stable = _is_stable(post_segment) + pre_mean = sum(pre_segment) / len(pre_segment) + post_mean = sum(post_segment) / len(post_segment) + + transition_width = abs(jump_end - jump_start) + 1 + shift = (pre_mean - post_mean) / pre_mean if pre_mean > 0 else 0 + + if ( + pre_stable + and post_stable + and shift > _REGRESSION_THRESHOLD + and transition_width <= 2 + ): + return "sudden_drop", (adj_left, adj_right) + + if ( + pre_stable + and post_stable + and shift > _REGRESSION_THRESHOLD + and transition_width > 2 + ): + decline_vals = daily_values[jump_start : jump_end + 1] + if len(decline_vals) >= 3: + x_vals = list(range(len(decline_vals))) + x_mean = sum(x_vals) / len(x_vals) + y_mean = sum(decline_vals) / len(decline_vals) + ss_xy = sum((x - x_mean) * (y - y_mean) for x, y in zip(x_vals, decline_vals)) + ss_xx = sum((x - x_mean) ** 2 for x in x_vals) + ss_yy = sum((y - y_mean) ** 2 for y in decline_vals) + if ss_xx > 0 and ss_yy > 0: + slope = ss_xy / ss_xx + r_squared = (ss_xy**2) / (ss_xx * ss_yy) + if slope < 0 and r_squared > 0.7: + return "gradual_decline", (adj_left, adj_right) + + return "other_reasons", (jump_start, jump_end) + + return "other_reasons", None + + +def classify_single_metric(daily_values, baseline, threshold=_REGRESSION_THRESHOLD): + """Two-step classification for one metric's time series. + + Step 1 -- Regression check: + Is the recent average more than ``threshold`` below the baseline? + If **no** -> ``"no_regression"``. + + Step 2 -- Regression subtype (only when Step 1 says *yes*): + Classify into one of ``"significant_fluctuation"``, + ``"occasional_spike"``, ``"sudden_drop"``, + ``"gradual_decline"``, or ``"other_reasons"``. + + Returns: + (curve_type, jump_interval) where jump_interval is + (start_index, end_index) or None. + """ + if not daily_values: + return "no_regression", None + + if not _is_regression(daily_values, baseline, threshold): + return "no_regression", None + + regression_type, jump_interval = _classify_regression_type(daily_values) + return regression_type, jump_interval + + +def _get_threshold_for_metric(baseline_data_list, metric): + """Get the pre-merge threshold for a metric from the latest baseline data. + + Looks for d_threshold_pre_merge_{metric_suffix} in the latest baseline + entry. Returns DEFAULT_THRESHOLD (5%) if not found. + """ + if not baseline_data_list: + return DEFAULT_THRESHOLD + latest_baseline = baseline_data_list[-1] + metric_suffix = metric[2:] # Remove "d_" prefix + threshold_key = f"d_threshold_pre_merge_{metric_suffix}" + if threshold_key in latest_baseline: + return latest_baseline[threshold_key] + return DEFAULT_THRESHOLD + + +def classify_test_case(grouped_data): + """Run classification on all metrics and aggregate results. + + Uses threshold from baseline data for each metric. Reads pre-computed + ``daily_data`` and ``baselines`` from each entry (populated by + :func:`get_baseline`) and stores classification results back into + ``grouped_data[key]``: + "curve_type": str (overall) + "per_metric_info": {metric: {"curve_type": str, + "jump_interval": (date_str, date_str) or None, + "jump_commits": {...} or None}} + """ + for key, bucket in grouped_data.items(): + daily_data = bucket.get("daily_data", {}) + baselines = bucket.get("baselines", {}) + baseline_data_list = bucket.get("baseline_data", []) + per_metric_results = {} + per_metric_info = {} + + for metric in CHART_METRICS: + md = daily_data.get(metric, {}) + daily_vals = md.get("values", []) + daily_dates = md.get("dates", []) + daily_entries = md.get("entries", []) + baseline = baselines.get(metric, 0.0) + + threshold = _get_threshold_for_metric(baseline_data_list, metric) + curve_type, jump = classify_single_metric(daily_vals, baseline, threshold) + per_metric_results[metric] = curve_type + + jump_dates = None + jump_commits = None + if jump is not None and daily_dates: + js, je = jump + js = max(0, min(js, len(daily_dates) - 1)) + je = max(0, min(je, len(daily_dates) - 1)) + jump_dates = (daily_dates[js], daily_dates[je]) + if curve_type in ("sudden_drop", "gradual_decline", "other_reasons"): + jump_commits = _extract_jump_commits(daily_entries, daily_dates, js, je) + + per_metric_info[metric] = { + "curve_type": curve_type, + "jump_interval": jump_dates, + "jump_commits": jump_commits, + } + + # Aggregate overall type using only CLASSIFICATION_METRICS. + # Both NR and OS are "transparent" (defer to the other metric). + # Priority: SF > OR > GD > SD > OS > NR + classification_types = [ + per_metric_results[m] for m in CLASSIFICATION_METRICS if m in per_metric_results + ] + + if not classification_types: + overall = "no_regression" + elif len(classification_types) == 1: + overall = classification_types[0] + else: + # 6x6 aggregation: merge two types via priority, where NR and + # OS are transparent (defer to the other curve's type). + _PRIORITY = { + "significant_fluctuation": 5, + "other_reasons": 4, + "gradual_decline": 3, + "sudden_drop": 2, + "occasional_spike": 1, + "no_regression": 0, + } + a, b = classification_types[0], classification_types[1] + pa, pb = _PRIORITY.get(a, 0), _PRIORITY.get(b, 0) + overall = a if pa >= pb else b + + bucket["curve_type"] = overall + bucket["per_metric_info"] = per_metric_info + + +# --------------------------------------------------------------------------- +# OpenSearch query + grouping +# --------------------------------------------------------------------------- + + +def get_history_data(extra_must_clauses=None): + """Query perf data from OpenSearch and group by (s_test_case_name, s_gpu_type). + + Queries both baseline and non-baseline data from the last + QUERY_LOOKBACK_DAYS days. Additional filters can be passed via + *extra_must_clauses*. + + Returns: + dict mapping (test_case, gpu_type) -> { + "history_data": [non-baseline entries sorted by time], + "baseline_data": [baseline entries sorted by time], + } + or None on query failure. + """ + must_clauses = [ + {"term": {"b_is_valid": True}}, + { + "range": { + "ts_created": { + "gte": int(time.time() - 24 * 3600 * QUERY_LOOKBACK_DAYS) + // (24 * 3600) + * 24 + * 3600 + * 1000, + } + } + }, + ] + if extra_must_clauses: + must_clauses.extend(extra_must_clauses) + + data_list = OpenSearchDB.queryPerfDataFromOpenSearchDB( + PERF_SANITY_PROJECT_NAME, must_clauses, size=MAX_QUERY_SIZE + ) + + if data_list is None: + return None + + groups = {} + for data in data_list: + key = ( + data.get("s_test_case_name", ""), + data.get("s_gpu_type", ""), + ) + groups.setdefault(key, {"history_data": [], "baseline_data": []}) + if data.get("b_is_baseline"): + groups[key]["baseline_data"].append(data) + else: + groups[key]["history_data"].append(data) + + for key, bucket in groups.items(): + bucket["history_data"] = sorted( + bucket["history_data"], + key=lambda d: _parse_timestamp(d.get("ts_created") or d.get("@timestamp", 0)), + ) + bucket["baseline_data"] = sorted( + bucket["baseline_data"], + key=lambda d: _parse_timestamp(d.get("ts_created") or d.get("@timestamp", 0)), + ) + + return groups + + +# --------------------------------------------------------------------------- +# SVG chart generation +# --------------------------------------------------------------------------- + +_SVG_WIDTH = 620 +_SVG_HEIGHT = 280 +_MARGIN = {"top": 30, "right": 20, "bottom": 55, "left": 75} +_PLOT_W = _SVG_WIDTH - _MARGIN["left"] - _MARGIN["right"] +_PLOT_H = _SVG_HEIGHT - _MARGIN["top"] - _MARGIN["bottom"] + + +def _generate_svg_chart( + history_points, + metric, + label, + new_points=None, + baseline_value=None, + threshold_line_value=None, + curve_type=None, + jump_interval=None, +): + """Return an SVG string for a single metric chart. + + Args: + history_points: list of (datetime, value) or (datetime, value, data_dict) + sorted by date. + metric: metric key string. + label: display label for the chart title. + new_points: optional list of (datetime, value) for new data (red dots). + baseline_value: optional float drawn as a horizontal dashed red line. + threshold_line_value: optional float drawn as a horizontal dashed + orange line (regression threshold). + curve_type: optional str -- the regression classification for this + metric (used for badge display). + jump_interval: optional (start_date_str, end_date_str) -- regression + window shading. + """ + all_values = [v for _, v, *_ in history_points if v is not None] + if new_points: + all_values.extend(v for _, v in new_points if v is not None) + if baseline_value is not None: + all_values.append(baseline_value) + if threshold_line_value is not None: + all_values.append(threshold_line_value) + + if not history_points and not new_points and baseline_value is None: + return f'
No data for {escape_html(label)}
' + if not all_values: + return ( + f'
No numeric data for {escape_html(label)}
' + ) + + min_val = min(all_values) + max_val = max(all_values) + val_range = max_val - min_val if max_val != min_val else 1.0 + min_val -= val_range * 0.05 + max_val += val_range * 0.05 + val_range = max_val - min_val + + dates = [d for d, *_ in history_points] + if new_points: + dates.extend(d for d, _ in new_points) + if not dates: + return ( + f'
No data points for {escape_html(label)}
' + ) + + min_ts = min(dates).timestamp() + max_ts = max(dates).timestamp() + ts_range = max_ts - min_ts if max_ts != min_ts else 1.0 + + def _x(dt): + return _MARGIN["left"] + (dt.timestamp() - min_ts) / ts_range * _PLOT_W + + def _x_date_str(date_str): + dt = datetime.strptime(date_str, "%Y-%m-%d") + ts = dt.timestamp() + ts = max(min_ts, min(ts, max_ts)) + return _MARGIN["left"] + (ts - min_ts) / ts_range * _PLOT_W + + def _y(v): + return _MARGIN["top"] + _PLOT_H - (v - min_val) / val_range * _PLOT_H + + svg = [ + f'' + ] + + # Grid lines (Y axis, 5 ticks) + for i in range(6): + v = min_val + val_range * i / 5 + y = _y(v) + svg.append( + f'' + ) + svg.append( + f'{v:.1f}' + ) + + # Jump interval shaded region + if jump_interval is not None: + j_start, j_end = jump_interval + jx1 = _x_date_str(j_start) + jx2 = _x_date_str(j_end) + if jx2 - jx1 < 4: + jx2 = jx1 + 4 + svg.append( + f'' + ) + svg.append( + f'' + ) + svg.append( + f'' + ) + + # Axes + svg.append( + f'' + ) + svg.append( + f'' + ) + + # X-axis date labels + unique_dates = sorted(set(dates)) + n_labels = min(6, len(unique_dates)) + if len(unique_dates) >= n_labels: + label_dates = unique_dates[:: max(1, len(unique_dates) // n_labels)][:n_labels] + else: + label_dates = unique_dates + for dt in label_dates: + x = _x(dt) + y_base = _MARGIN["top"] + _PLOT_H + svg.append( + f'{dt.strftime("%m/%d")}' + ) + + # Title with curve type badge + title_text = escape_html(label) + svg.append( + f'{title_text}' + ) + if curve_type and curve_type != "no_regression": + ct_color = _CURVE_TYPE_COLORS.get(curve_type, "#888") + ct_short = _CURVE_TYPE_LABELS.get(curve_type, curve_type) + badge_x = _SVG_WIDTH - _MARGIN["right"] - 4 + badge_y = _MARGIN["top"] - 16 + badge_text = ct_short + if jump_interval: + badge_text += f" [{jump_interval[0]} ~ {jump_interval[1]}]" + text_w = len(badge_text) * 5.5 + 10 + rx = badge_x - text_w + svg.append( + f'' + ) + svg.append( + f'{escape_html(badge_text)}' + ) + + # Baseline horizontal line (dashed red) + if baseline_value is not None: + by = _y(baseline_value) + svg.append( + f'' + ) + + # Threshold horizontal line (dashed orange) + if threshold_line_value is not None: + ty = _y(threshold_line_value) + svg.append( + f'' + ) + + # History line + dots (blue) + sorted_hist = sorted( + [(d, v, *rest) for d, v, *rest in history_points if v is not None], + key=lambda p: p[0], + ) + if len(sorted_hist) > 1: + path_d = " ".join( + f"{'M' if i == 0 else 'L'}{_x(d):.1f},{_y(v):.1f}" + for i, (d, v, *_) in enumerate(sorted_hist) + ) + svg.append(f'') + for item in sorted_hist: + d, v = item[0], item[1] + dd = item[2] if len(item) > 2 else None + if dd is not None: + json_attr = _data_dict_to_json_attr(dd) + svg.append( + f'' + f"{d.strftime('%Y-%m-%d %H:%M')} {v:.2f}" + ) + else: + svg.append(f'') + + # New data points (red) + if new_points: + for d, v in new_points: + if v is None: + continue + svg.append( + f'' + ) + + # Legend + legend_y = _MARGIN["top"] + _PLOT_H + 35 + legend_x = _MARGIN["left"] + 10 + svg.append(f'') + svg.append( + f'History' + ) + legend_x += 70 + if new_points: + svg.append(f'') + svg.append( + f'New' + ) + legend_x += 50 + if baseline_value is not None: + svg.append( + f'' + ) + svg.append( + f'Baseline ({baseline_value:.2f})' + ) + legend_x += 150 + if threshold_line_value is not None: + svg.append( + f'' + ) + svg.append( + f'Threshold ({threshold_line_value:.2f})' + ) + + svg.append("") + return "\n".join(svg) + + +# --------------------------------------------------------------------------- +# HTML report generation (post-merge dashboard) +# --------------------------------------------------------------------------- + + +def generate_post_merge_html(grouped_data, output_file): + """Generate a post-merge HTML dashboard from grouped perf data. + + This produces a full interactive report with three-way cascading filters + (GPU Type, Test Case, Curve Type), summary tables, and click-to-inspect + data-point popups. + """ + all_gpu_types = sorted(set(gpu for _, gpu in grouped_data.keys())) + all_test_cases = sorted(set(tc for tc, _ in grouped_data.keys())) + all_curve_types_set = set() + + sections = [] + section_tuples = [] + + for (test_case, gpu_type), bucket in sorted(grouped_data.items()): + history_data = bucket["history_data"] + curve_type = bucket.get("curve_type", "no_regression") + baselines = bucket.get("baselines", {}) + per_metric_info = bucket.get("per_metric_info", {}) + + all_curve_types_set.add(curve_type) + section_tuples.append((gpu_type, test_case, curve_type)) + + charts = [] + for metric in CHART_METRICS: + label = METRIC_LABELS.get(metric, metric) + hist_pts = _extract_points(history_data, metric) + baseline_val = baselines.get(metric) + m_info = per_metric_info.get(metric, {}) + charts.append( + _generate_svg_chart( + hist_pts, + metric, + label, + baseline_value=baseline_val, + curve_type=m_info.get("curve_type"), + jump_interval=m_info.get("jump_interval"), + ) + ) + + # Summary table + summary_rows = "" + if history_data: + latest = history_data[-1] + for metric in CHART_METRICS: + val = latest.get(metric) + bl_val = baselines.get(metric) + diff_str = "" + if val is not None and bl_val is not None and bl_val != 0: + diff_pct = (val - bl_val) / bl_val * 100 + color = "#0d904f" if diff_pct >= 0 else "#d93025" + diff_str = f' ({diff_pct:+.2f}%)' + val_str = f"{val:.2f}" if val is not None else "N/A" + bl_str = f"{bl_val:.2f}" if bl_val is not None else "N/A" + m_info = per_metric_info.get(metric, {}) + m_ct = m_info.get("curve_type", "no_regression") + m_ct_color = _CURVE_TYPE_COLORS.get(m_ct, "#888") + m_ct_label = _CURVE_TYPE_LABELS.get(m_ct, m_ct) + m_jump = m_info.get("jump_interval") + jump_str = "" + if m_jump: + jump_str = ( + f' ' + f"[{m_jump[0]} ~ {m_jump[1]}]" + ) + ct_cell = ( + f'{m_ct_label}' + f"{jump_str}" + ) + jc = m_info.get("jump_commits") + jl_cell = "" + jr_cell = "" + if jc: + left = jc.get("left") + right = jc.get("right") + if left and left.get("s_commit"): + short = left["s_commit"][:8] + ts = left.get("timestamp", "") + jl_cell = ( + f"{escape_html(short)}" + f'
' + f"{escape_html(ts)}" + ) + if right and right.get("s_commit"): + short = right["s_commit"][:8] + ts = right.get("timestamp", "") + jr_cell = ( + f"{escape_html(short)}" + f'
' + f"{escape_html(ts)}" + ) + summary_rows += ( + f"{METRIC_LABELS.get(metric, metric)}" + f"{val_str}{diff_str}" + f"{bl_str}" + f"{ct_cell}" + f"{jl_cell}" + f"{jr_cell}" + ) + + n_points = len(history_data) + ct_color = _CURVE_TYPE_COLORS.get(curve_type, "#888") + ct_label = _CURVE_TYPE_LABELS.get(curve_type, curve_type) + + header = escape_html(f"{test_case} [{gpu_type}]") + data_gpu = escape_html(gpu_type) + data_test = escape_html(test_case) + data_curve = escape_html(curve_type) + table_header = ( + "MetricLatest Value" + "Baseline (P95)Curve Type" + "Jump LeftJump Right" + ) + section = f""" +
+ {header} + {n_points} runs + {ct_label} + +
+ {"".join(charts)} +
+ { + "" + if not summary_rows + else f''' + + {table_header} + {summary_rows} +
+ ''' + } +
+ """ + sections.append(section) + + all_curve_types = sorted(all_curve_types_set) + + gpu_to_tests = {} + test_to_gpus = {} + for tc, gpu in grouped_data.keys(): + gpu_to_tests.setdefault(gpu, []) + if tc not in gpu_to_tests[gpu]: + gpu_to_tests[gpu].append(tc) + test_to_gpus.setdefault(tc, []) + if gpu not in test_to_gpus[tc]: + test_to_gpus[tc].append(gpu) + for k in gpu_to_tests: + gpu_to_tests[k].sort() + for k in test_to_gpus: + test_to_gpus[k].sort() + + triples_json = _json.dumps(section_tuples) + + gpu_chips = [ + '' + ] + for gpu in all_gpu_types: + gpu_chips.append( + f'' + ) + + test_chips = [ + '' + ] + for tc in all_test_cases: + test_chips.append( + f'' + ) + + curve_chips = [ + '' + ] + for ct in all_curve_types: + ct_color = _CURVE_TYPE_COLORS.get(ct, "#888") + ct_label = _CURVE_TYPE_LABELS.get(ct, ct) + curve_chips.append( + f'" + ) + + html = f""" + + + + Perf Sanity History Dashboard + + + +

Perf Sanity History Dashboard

+

+ {len(grouped_data)} test case(s) · + Lookback: {QUERY_LOOKBACK_DAYS} days · + Generated: {datetime.now().strftime("%Y-%m-%d %H:%M:%S")} +

+ +
+

GPU Type

+
+ {"".join(gpu_chips)} +
+

Test Case

+
+ {"".join(test_chips)} +
+

Curve Type

+
+ {"".join(curve_chips)} +
+
+
+ +
+ + +
+ +
+ {"".join(sections)} +
+ + + + +""" + with open(output_file, "w", encoding="utf-8") as f: + f.write(html) + print(f"Generated perf history report with {len(grouped_data)} test cases: {output_file}") diff --git a/tensorrt_llm/_torch/pyexecutor/py_executor.py b/tensorrt_llm/_torch/pyexecutor/py_executor.py index 1f4049ebddb3..cbd1d7e91a46 100644 --- a/tensorrt_llm/_torch/pyexecutor/py_executor.py +++ b/tensorrt_llm/_torch/pyexecutor/py_executor.py @@ -2482,7 +2482,10 @@ def _fetch_new_requests( if self.enable_iter_perf_stats and self.dist.rank == 0: self._update_new_active_requests_queue_latency(new_requests) - # 5. Schedule requests across ranks (DP only) + # 5. Update total fetch counter (used by benchmark fill loop) + self.num_fetch_requests += len(new_requests) + + # 6. Schedule requests across ranks (DP only) if self.enable_attention_dp: all_ranks_new_requests, self.expected_num_active_requests = \ self.adp_router.route_requests( @@ -2490,13 +2493,12 @@ def _fetch_new_requests( self.max_num_active_requests) new_requests_cur_rank = all_ranks_new_requests[self.dist.tp_rank] - # Update counters for DP - self.num_fetch_requests += len(new_requests) + # Update per-rank counter for DP self.num_fetch_requests_cur_rank += len(new_requests_cur_rank) new_requests = new_requests_cur_rank - # 6. Merge requests + # 7. Merge requests return merge_requests(new_requests, cp_config=self.dist.cp_config, cp_rank=self.dist.cp_rank, diff --git a/tests/integration/defs/perf/open_search_db_utils.py b/tests/integration/defs/perf/open_search_db_utils.py index 53d849bc479a..9b7f7895ace8 100644 --- a/tests/integration/defs/perf/open_search_db_utils.py +++ b/tests/integration/defs/perf/open_search_db_utils.py @@ -243,16 +243,76 @@ def is_empty(value): return True -def calculate_best_perf_result(history_data_list, new_data): +def _rolling_smooth(values, window=3): + """Trailing rolling mean with same-length output. + + Early elements use fewer samples (i.e. the first element is itself, + the second is the mean of the first two, etc.). """ - Get the best performance metrics from history data and new data + if not values: + return [] + smoothed = [] + for i in range(len(values)): + start = max(0, i - window + 1) + w = values[start:i + 1] + smoothed.append(sum(w) / len(w)) + return smoothed + + +def _percentile(values, p): + """Compute the p-th percentile with linear interpolation.""" + if not values: + return 0.0 + s = sorted(values) + k = (p / 100.0) * (len(s) - 1) + lo = int(k) + hi = min(lo + 1, len(s) - 1) + frac = k - lo + return s[lo] + frac * (s[hi] - s[lo]) + + +def _daily_aggregate_values(data_list, metric): + """Aggregate multiple data points on the same day to a single mean value. + + Returns a list of daily-aggregated metric values sorted by date. + """ + by_day = {} + for data in data_list: + if data.get("b_is_baseline"): + continue + val = data.get(metric) + if val is None: + continue + ts = data.get("ts_created") or data.get("@timestamp") + if ts is None: + continue + if isinstance(ts, (int, float)): + if ts > 1e12: + ts = ts / 1000 + day_key = datetime.fromtimestamp(ts).strftime("%Y-%m-%d") + elif isinstance(ts, datetime): + day_key = ts.strftime("%Y-%m-%d") + else: + day_key = str(ts)[:10] + by_day.setdefault(day_key, []).append(val) + result = [] + for day in sorted(by_day): + vals = by_day[day] + result.append(sum(vals) / len(vals)) + return result + + +def calculate_baseline_metrics(history_data_list, new_data): + """Calculate baseline metrics using rolling smooth + percentile algorithm. + + For each metric, aggregates data to daily values, applies a trailing + rolling mean (window=3), then takes: + - P95 for MAXIMIZE_METRICS (larger is better, e.g. throughput) + - P5 for MINIMIZE_METRICS (smaller is better, e.g. latency) """ - # Combine history data and new data all_data = [] if history_data_list: all_data.extend(history_data_list) - - # Handle new_data as either a single dict or list if isinstance(new_data, list): all_data.extend(new_data) elif new_data: @@ -261,35 +321,18 @@ def calculate_best_perf_result(history_data_list, new_data): if not all_data: return {} - best_metrics = {} - - # Calculate best values for maximize metrics - for metric in MAXIMIZE_METRICS: - values = [] - for data in all_data: - # Skip baseline data - if data.get("b_is_baseline") and data.get("b_is_baseline") == True: - continue - if metric not in data: - continue - values.append(data.get(metric)) - if values: - best_metrics[metric] = max(values) - - # Calculate best values for minimize metrics - for metric in MINIMIZE_METRICS: - values = [] - for data in all_data: - # Skip baseline data - if data.get("b_is_baseline") and data.get("b_is_baseline") == True: - continue - if metric not in data: - continue - values.append(data.get(metric)) - if values: - best_metrics[metric] = min(values) + baseline_metrics = {} + for metric in MAXIMIZE_METRICS + MINIMIZE_METRICS: + daily_vals = _daily_aggregate_values(all_data, metric) + if not daily_vals: + continue + smoothed = _rolling_smooth(daily_vals, window=3) + if metric in MAXIMIZE_METRICS: + baseline_metrics[metric] = _percentile(smoothed, 95) + else: + baseline_metrics[metric] = _percentile(smoothed, 5) - return best_metrics + return baseline_metrics def get_history_data(new_data_dict, match_keys, common_values_dict): @@ -349,11 +392,6 @@ def parse_timestamp(timestamp): "b_is_post_merge": True } }, - { - "term": { - "b_is_regression": False - } - }, { "range": { "ts_created": { @@ -510,8 +548,8 @@ def prepare_baseline_data(history_baseline_dict, history_data_dict, cmd_idxs = new_data_dict.keys() # Find the best history post-merge data for each cmd for cmd_idx in cmd_idxs: - # Calculate best metrics from history post-merge data and new data - best_metrics = calculate_best_perf_result(history_data_dict[cmd_idx], + # Calculate baseline metrics using rolling smooth + P95 algorithm + best_metrics = calculate_baseline_metrics(history_data_dict[cmd_idx], new_data_dict[cmd_idx]) # Create new_baseline_data from new_data_dict and set b_is_baseline @@ -590,94 +628,115 @@ def _get_metric_keys(): return metric_keys -def _print_regression_data(data, print_func=None): +def generate_perf_yaml(new_data_dict, output_dir=None): """ - Print regression info and config. - """ - if print_func is None: - print_func = print_info - - if "s_regression_info" in data: - print_func("=== Regression Info ===") - for item in data["s_regression_info"].split(","): - print_func(item.strip()) - - metric_keys = _get_metric_keys() + Save new perf data entries to perf_data.yaml for post-processing. - print_func("\n=== Config ===") - config_keys = sorted([key for key in data.keys() if key not in metric_keys]) - for key in config_keys: - if key == "s_regression_info": - continue - value = data[key] - print_func(f'"{key}": {value}') - - -def check_perf_regression(new_data_dict, - fail_on_regression=False, - output_dir=None): + Each entry in the output list is a dict with: + - "new_data": the new perf data dict """ - Check performance regression by printing regression data from new_data_dict. - If fail_on_regression is True, raises RuntimeError when regressions are found. - (This is a temporary feature to fail regression tests. We are observing the stability and will fail them by default soon.) - If output_dir is provided, saves regression data to regression_data.yaml. - """ - # Filter regression data from new_data_dict - regressive_data_list = [ - data for data in new_data_dict.values() - if data.get("b_is_regression", False) - ] - # Split regression data into post-merge and pre-merge - post_merge_regressions = [ - data for data in regressive_data_list - if data.get("b_is_post_merge", False) - ] - pre_merge_regressions = [ - data for data in regressive_data_list - if not data.get("b_is_post_merge", False) - ] - - # Save regression data to yaml file if output_dir is provided - if output_dir is not None and len(regressive_data_list) > 0: - regression_data_file = os.path.join(output_dir, "regression_data.yaml") - with open(regression_data_file, 'w') as f: - yaml.dump(regressive_data_list, f, default_flow_style=False) + all_entries = [] + for cmd_idx, new_data in new_data_dict.items(): + entry = {"new_data": new_data} + all_entries.append(entry) + + if output_dir is not None and len(all_entries) > 0: + perf_data_file = os.path.join(output_dir, "perf_data.yaml") + with open(perf_data_file, 'w') as f: + yaml.dump(all_entries, f, default_flow_style=False) print_info( - f"Saved {len(regressive_data_list)} regression data to {regression_data_file}" - ) - - # Print pre-merge regression data with print_warning - if len(pre_merge_regressions) > 0: - print_warning( - f"Found {len(pre_merge_regressions)} pre-merge perf regression data" - ) - for i, data in enumerate(pre_merge_regressions): - print_warning(f"\n{'=' * 60}") - print_warning(f"Pre-merge Regression Data #{i + 1}") - print_warning("=" * 60) - _print_regression_data(data, print_func=print_warning) - - if fail_on_regression: - raise RuntimeError( - f"Found {len(pre_merge_regressions)} pre-merge perf regression data" - ) - - # Print post-merge regression data with print_warning - if len(post_merge_regressions) > 0: - print_warning( - f"Found {len(post_merge_regressions)} post-merge perf regression data" - ) - for i, data in enumerate(post_merge_regressions): - print_warning(f"\n{'=' * 60}") - print_warning(f"Post-merge Regression Data #{i + 1}") - print_warning("=" * 60) - _print_regression_data(data, print_func=print_warning) - - if fail_on_regression: - raise RuntimeError( - f"Found {len(post_merge_regressions)} post-merge perf regression data" - ) - - # Print summary if no regressions - if len(regressive_data_list) == 0: - print_info("No regression data found.") + f"Saved {len(all_entries)} perf data entries to {perf_data_file}") + elif len(all_entries) == 0: + print_info("No perf data to save.") + + +# def _print_regression_data(data, print_func=None): +# """ +# Print regression info and config. +# """ +# if print_func is None: +# print_func = print_info +# +# if "s_regression_info" in data: +# print_func("=== Regression Info ===") +# for item in data["s_regression_info"].split(","): +# print_func(item.strip()) +# +# metric_keys = _get_metric_keys() +# +# print_func("\n=== Config ===") +# config_keys = sorted([key for key in data.keys() if key not in metric_keys]) +# for key in config_keys: +# if key == "s_regression_info": +# continue +# value = data[key] +# print_func(f'"{key}": {value}') + +# def check_perf_regression(new_data_dict, +# fail_on_regression=False, +# output_dir=None): +# """ +# Check performance regression by printing regression data from new_data_dict. +# If fail_on_regression is True, raises RuntimeError when regressions are found. +# (This is a temporary feature to fail regression tests. We are observing the stability and will fail them by default soon.) +# If output_dir is provided, saves regression data to regression_data.yaml. +# """ +# # Filter regression data from new_data_dict +# regressive_data_list = [ +# data for data in new_data_dict.values() +# if data.get("b_is_regression", False) +# ] +# # Split regression data into post-merge and pre-merge +# post_merge_regressions = [ +# data for data in regressive_data_list +# if data.get("b_is_post_merge", False) +# ] +# pre_merge_regressions = [ +# data for data in regressive_data_list +# if not data.get("b_is_post_merge", False) +# ] +# +# # Save regression data to yaml file if output_dir is provided +# if output_dir is not None and len(regressive_data_list) > 0: +# regression_data_file = os.path.join(output_dir, "regression_data.yaml") +# with open(regression_data_file, 'w') as f: +# yaml.dump(regressive_data_list, f, default_flow_style=False) +# print_info( +# f"Saved {len(regressive_data_list)} regression data to {regression_data_file}" +# ) +# +# # Print pre-merge regression data with print_warning +# if len(pre_merge_regressions) > 0: +# print_warning( +# f"Found {len(pre_merge_regressions)} pre-merge perf regression data" +# ) +# for i, data in enumerate(pre_merge_regressions): +# print_warning(f"\n{'=' * 60}") +# print_warning(f"Pre-merge Regression Data #{i + 1}") +# print_warning("=" * 60) +# _print_regression_data(data, print_func=print_warning) +# +# if fail_on_regression: +# raise RuntimeError( +# f"Found {len(pre_merge_regressions)} pre-merge perf regression data" +# ) +# +# # Print post-merge regression data with print_warning +# if len(post_merge_regressions) > 0: +# print_warning( +# f"Found {len(post_merge_regressions)} post-merge perf regression data" +# ) +# for i, data in enumerate(post_merge_regressions): +# print_warning(f"\n{'=' * 60}") +# print_warning(f"Post-merge Regression Data #{i + 1}") +# print_warning("=" * 60) +# _print_regression_data(data, print_func=print_warning) +# +# if fail_on_regression: +# raise RuntimeError( +# f"Found {len(post_merge_regressions)} post-merge perf regression data" +# ) +# +# # Print summary if no regressions +# if len(regressive_data_list) == 0: +# print_info("No regression data found.") diff --git a/tests/integration/defs/perf/test_perf_sanity.py b/tests/integration/defs/perf/test_perf_sanity.py index 3dcd2be5fc83..cef22a01b749 100644 --- a/tests/integration/defs/perf/test_perf_sanity.py +++ b/tests/integration/defs/perf/test_perf_sanity.py @@ -35,7 +35,7 @@ from .open_search_db_utils import ( SCENARIO_MATCH_FIELDS, add_id, - check_perf_regression, + generate_perf_yaml, get_common_values, get_history_data, get_job_info, @@ -66,8 +66,10 @@ } DEFAULT_TIMEOUT = 5400 -AGGR_CONFIG_FOLDER = "tests/scripts/perf-sanity" -DISAGG_CONFIG_FOLDER = "tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity" +AGG_CONFIG_FOLDER = os.environ.get("AGG_CONFIG_FOLDER", "tests/scripts/perf-sanity/aggregated") +DISAGG_CONFIG_FOLDER = os.environ.get( + "DISAGG_CONFIG_FOLDER", "tests/scripts/perf-sanity/disaggregated" +) # Regex patterns for parsing benchmark output metrics # Key is the metric name used in database (e.g., "mean_e2el", "seq_throughput") @@ -901,12 +903,16 @@ def get_config_dir(benchmark_mode: Optional[str]) -> str: benchmark_mode: "e2e", "gen_only", "ctx_only", or None (for normal aggr) Returns: - str: Config directory path (relative to llm_root) + str: Absolute config directory path """ if benchmark_mode in ("e2e", "gen_only", "ctx_only"): - return DISAGG_CONFIG_FOLDER + config_folder = DISAGG_CONFIG_FOLDER else: - return AGGR_CONFIG_FOLDER + config_folder = AGG_CONFIG_FOLDER + # If relative path, join with llm root + if not os.path.isabs(config_folder): + config_folder = os.path.join(get_llm_root(), config_folder) + return config_folder class PerfSanityTestConfig: @@ -962,10 +968,7 @@ def get_gpu_type() -> str: ) # Get config_dir based on benchmark_mode - config_dir = get_config_dir(self.benchmark_mode) - self.config_dir = os.getenv( - "TRTLLM_CONFIG_FOLDER", os.path.join(get_llm_root(), config_dir) - ) + self.config_dir = get_config_dir(self.benchmark_mode) def parse_config_file(self): """Parse config file based on runtime and benchmark_mode.""" @@ -1425,7 +1428,7 @@ def add_dict_prefix(config_dict: dict, prefix_name: str) -> dict: if not match_keys: if server_config.match_mode == "scenario": match_keys = SCENARIO_MATCH_FIELDS.copy() - is_scenario_mode = True + is_scenario_mode = True # noqa: F841 else: match_keys.extend(["s_gpu_type", "s_runtime"]) match_keys.extend(server_config.to_match_keys()) @@ -1538,11 +1541,12 @@ def add_dict_prefix(config_dict: dict, prefix_name: str) -> dict: # Upload the new perf data and baseline data to database post_new_perf_data(new_baseline_data_dict, new_data_dict) - check_perf_regression( + generate_perf_yaml( new_data_dict, - fail_on_regression=is_scenario_mode, output_dir=self.test_output_dir, ) + # TODO: Re-enable regression failure check if needed + # check_perf_regression(new_data_dict, fail_on_regression=is_scenario_mode, output_dir=self.test_output_dir) # Perf sanity test case parameters @@ -1575,8 +1579,10 @@ def get_yaml_files_with_server_names(directory: str) -> Dict[str, List[str]]: def get_aggr_test_cases() -> List[str]: """Generate aggr test cases based on actual server_config names in YAML files.""" - llm_root = get_llm_root() - aggr_config_dir = os.path.join(llm_root, AGGR_CONFIG_FOLDER) + aggr_config_dir = AGG_CONFIG_FOLDER + # If relative path, join with llm root + if not os.path.isabs(aggr_config_dir): + aggr_config_dir = os.path.join(get_llm_root(), aggr_config_dir) yaml_server_names = get_yaml_files_with_server_names(aggr_config_dir) test_cases = [] @@ -1593,15 +1599,11 @@ def get_aggr_test_cases() -> List[str]: def get_disagg_test_cases() -> List[str]: - """Generate disagg test cases with benchmark modes. - - New format: - - Disagg e2e: {test_type}-e2e-{config_base} - - Disagg gen_only: {test_type}-gen_only-{config_base} - - ctx_only: aggr_{upload}-ctx_only-{config_base} (uses aggr prefix) - """ - llm_root = get_llm_root() - disagg_config_dir = os.path.join(llm_root, DISAGG_CONFIG_FOLDER) + """Generate disagg test cases with benchmark modes.""" + disagg_config_dir = DISAGG_CONFIG_FOLDER + # If relative path, join with llm root + if not os.path.isabs(disagg_config_dir): + disagg_config_dir = os.path.join(get_llm_root(), disagg_config_dir) yaml_files = glob.glob(os.path.join(disagg_config_dir, "*.yaml")) basenames = sorted([os.path.splitext(os.path.basename(f))[0] for f in yaml_files]) diff --git a/tests/integration/test_lists/test-db/l0_gb200_multi_nodes_perf_sanity_ctx1_node1_gpu1_gen1_node1_gpu2.yml b/tests/integration/test_lists/test-db/l0_gb200_multi_nodes_perf_sanity_ctx1_node1_gpu1_gen1_node1_gpu2.yml index c551a6ce311f..046b23572816 100644 --- a/tests/integration/test_lists/test-db/l0_gb200_multi_nodes_perf_sanity_ctx1_node1_gpu1_gen1_node1_gpu2.yml +++ b/tests/integration/test_lists/test-db/l0_gb200_multi_nodes_perf_sanity_ctx1_node1_gpu1_gen1_node1_gpu2.yml @@ -16,7 +16,7 @@ l0_gb200_multi_nodes_perf_sanity_ctx1_node1_gpu1_gen1_node1_gpu2: tests: - perf/test_perf_sanity.py::test_e2e[disagg_upload-gen_only-gb200_gpt-oss-120b-fp4_1k1k_con2048_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-UCX] TIMEOUT (120) - perf/test_perf_sanity.py::test_e2e[disagg_upload-gen_only-gb200_gpt-oss-120b-fp4_1k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-UCX] TIMEOUT (120) - - perf/test_perf_sanity.py::test_e2e[disagg_upload-gen_only-gb200_gpt-oss-120b-fp4_8k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-UCX] TIMEOUT (120) + # - perf/test_perf_sanity.py::test_e2e[disagg_upload-gen_only-gb200_gpt-oss-120b-fp4_8k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-UCX] TIMEOUT (120) # - perf/test_perf_sanity.py::test_e2e[disagg_upload-e2e-gb200_gpt-oss-120b-fp4_1k1k_con2048_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-UCX] TIMEOUT (120) # - perf/test_perf_sanity.py::test_e2e[disagg_upload-e2e-gb200_gpt-oss-120b-fp4_1k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-UCX] TIMEOUT (120) # - perf/test_perf_sanity.py::test_e2e[disagg_upload-e2e-gb200_gpt-oss-120b-fp4_8k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-UCX] TIMEOUT (120) diff --git a/tests/integration/test_lists/test-db/l0_gb200_multi_nodes_perf_sanity_ctx1_node1_gpu1_gen1_node1_gpu4.yml b/tests/integration/test_lists/test-db/l0_gb200_multi_nodes_perf_sanity_ctx1_node1_gpu1_gen1_node1_gpu4.yml index 72c189df6630..3037fef728a4 100644 --- a/tests/integration/test_lists/test-db/l0_gb200_multi_nodes_perf_sanity_ctx1_node1_gpu1_gen1_node1_gpu4.yml +++ b/tests/integration/test_lists/test-db/l0_gb200_multi_nodes_perf_sanity_ctx1_node1_gpu1_gen1_node1_gpu4.yml @@ -17,7 +17,7 @@ l0_gb200_multi_nodes_perf_sanity_ctx1_node1_gpu1_gen1_node1_gpu4: - perf/test_perf_sanity.py::test_e2e[disagg_upload-gen_only-gb200_gpt-oss-120b-fp4_1k1k_con64_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-UCX] TIMEOUT (120) - perf/test_perf_sanity.py::test_e2e[disagg_upload-gen_only-gb200_gpt-oss-120b-fp4_8k1k_con128_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-UCX] TIMEOUT (120) - perf/test_perf_sanity.py::test_e2e[disagg_upload-gen_only-gb200_gpt-oss-120b-fp4_8k1k_con4_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-UCX] TIMEOUT (120) - - perf/test_perf_sanity.py::test_e2e[disagg_upload-gen_only-gb200_qwen3-235b-fp4_8k1k_con64_ctx1_tp1_gen1_tep4_eplb0_mtp0_ccb-UCX] TIMEOUT (120) + # - perf/test_perf_sanity.py::test_e2e[disagg_upload-gen_only-gb200_qwen3-235b-fp4_8k1k_con64_ctx1_tp1_gen1_tep4_eplb0_mtp0_ccb-UCX] TIMEOUT (120) # - perf/test_perf_sanity.py::test_e2e[disagg_upload-e2e-gb200_gpt-oss-120b-fp4_1k1k_con64_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-UCX] TIMEOUT (120) # - perf/test_perf_sanity.py::test_e2e[disagg_upload-e2e-gb200_gpt-oss-120b-fp4_8k1k_con128_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-UCX] TIMEOUT (120) # - perf/test_perf_sanity.py::test_e2e[disagg_upload-e2e-gb200_gpt-oss-120b-fp4_8k1k_con4_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-UCX] TIMEOUT (120) diff --git a/tests/integration/test_lists/waives.txt b/tests/integration/test_lists/waives.txt index 47c5897bba43..39509f843af3 100644 --- a/tests/integration/test_lists/waives.txt +++ b/tests/integration/test_lists/waives.txt @@ -285,8 +285,6 @@ accuracy/test_llm_api_pytorch.py::TestQwen3_30B_A3B::test_fp8[latency-torch_comp accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_w4_4gpus[v1_kv_cache-dp4-trtllm-auto] SKIP (https://nvbugs/5596343) test_e2e.py::test_trtllm_multimodal_benchmark_serving SKIP (https://nvbugs/5864769) unittest/_torch/auto_deploy/unit/multigpu/transformations/library/test_bmm_sharding.py::test_sharding[1-1] SKIP (https://nvbugs/5875203) -perf/test_perf_sanity.py::test_e2e[aggr_upload-k2_thinking_fp4_2_nodes_grace_blackwell-k2_thinking_fp4_dep8_32k8k] SKIP (https://nvbugs/5846166) -perf/test_perf_sanity.py::test_e2e[aggr_upload-k2_thinking_fp4_2_nodes_grace_blackwell-k2_thinking_fp4_tep8_32k8k] SKIP (https://nvbugs/5846166) accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_fp8_block_scales[mtp=vanilla-fp8kv=False-attention_dp=False-cuda_graph=False-overlap_scheduler=False-torch_compile=False] SKIP (https://nvbugs/5879577) accuracy/test_llm_api_pytorch.py::TestMiniMaxM2::test_4gpus[attention_dp=False-cuda_graph=True-overlap_scheduler=True-tp_size=4-ep_size=4] SKIP (https://nvbugs/5879588) accuracy/test_llm_api_pytorch.py::TestNemotronV3Super::test_auto_dtype_4gpus[4-1-False-False-False] SKIP (https://nvbugs/5879625) @@ -305,7 +303,6 @@ accuracy/test_llm_api_pytorch.py::TestNemotronV3Super::test_fp8_4gpus[attention_ accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_eagle3_4gpus[v1_kv_cache-cutlass-one_model-no_overlap_scheduler] SKIP (https://nvbugs/5809169) accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_eagle3_4gpus[v2_kv_cache-cutlass-one_model-no_overlap_scheduler] SKIP (https://nvbugs/5809169) accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_w4_4gpus[v2_kv_cache-dp4-trtllm-auto] SKIP (https://nvbugs/5888588) -perf/test_perf_sanity.py::test_e2e[aggr_upload-ctx_only-gb200_kimi-k2-thinking-fp4_8k1k_con4_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX] SKIP full:sm89/accuracy/test_disaggregated_serving.py::TestLlama3_1_8BInstruct::test_ngram SKIP (https://nvbugs/5893116) full:sm89/accuracy/test_disaggregated_serving.py::TestLlama3_1_8BInstruct::test_guided_decoding[xgrammar] SKIP (https://nvbugs/5893116) full:sm89/accuracy/test_disaggregated_serving.py::TestLlama3_1_8BInstruct::test_guided_decoding[llguidance] SKIP (https://nvbugs/5893116) @@ -338,23 +335,9 @@ unittest/_torch/modules/test_fused_moe.py::test_fused_moe_w4a8_nvfp4_fp8[TRTLLM] unittest/_torch/visual_gen/test_wan.py::TestWanTwoStageTransformer::test_two_stage_with_trtllm_attention SKIP (https://nvbugspro.nvidia.com/bug/5916830) disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_mpi[DeepSeek-V3-Lite-fp8] SKIP (https://nvbugs/5920761) accuracy/test_llm_api_pytorch.py::TestDeepSeekV32::test_fp8_blockscale[latency_default] SKIP (https://nvbugs/5920751) -perf/test_perf_sanity.py::test_e2e[disagg_upload-gen_only-gb200_deepseek-v32-fp4_1k1k_con2048_ctx1_dep4_gen1_dep4_eplb0_mtp1_ccb-UCX] SKIP (https://nvbugs/5846166) -perf/test_perf_sanity.py::test_e2e[disagg_upload-gen_only-gb200_kimi-k2-thinking-fp4_1k1k_con4_ctx1_dep4_gen1_tep4_eplb0_mtp0_ccb-UCX] SKIP (https://nvbugs/5846166) accuracy/test_llm_api_autodeploy.py::TestNemotronNanoV3::test_accuracy[fp8-1-trtllm] SKIP (https://nvbugs/5921674) cpp/test_unit_tests.py::test_unit_tests[kernels-80] SKIP (https://nvbugs/5924144) -perf/test_perf_sanity.py::test_e2e[aggr_upload-deepseek_r1_fp4_v2_grace_blackwell-r1_fp4_v2_dep4_mtp1_1k8k] SKIP (https://nvbugs/5846166) -perf/test_perf_sanity.py::test_e2e[disagg_upload-gen_only-gb200_deepseek-r1-fp4_1k1k_con1024_ctx1_dep4_gen1_dep8_eplb0_mtp0_ccb-UCX] SKIP (https://nvbugs/5846166) -perf/test_perf_sanity.py::test_e2e[disagg_upload-gen_only-gb200_qwen3-235b-fp4_8k1k_con64_ctx1_tp1_gen1_tep4_eplb0_mtp0_ccb-UCX] SKIP (https://nvbugs/5846166) full:RTXPro6000D/accuracy/test_llm_api_pytorch.py::TestQwen3_30B_A3B::test_nvfp4[dep4_latency_moe_cutlass-torch_compile=True] SKIP (https://nvbugs/5929339) -perf/test_perf_sanity.py::test_e2e[disagg_upload-gen_only-gb200_gpt-oss-120b-fp4_8k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-UCX] SKIP (https://nvbugs/5846166) -perf/test_perf_sanity.py::test_e2e[disagg_upload-gen_only-gb200_gpt-oss-120b-fp4_1k1k_con64_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-UCX] SKIP (https://nvbugs/5846166) -perf/test_perf_sanity.py::test_e2e[disagg_upload-gen_only-gb200_gpt-oss-120b-fp4_8k1k_con128_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-UCX] SKIP (https://nvbugs/5846166) -perf/test_perf_sanity.py::test_e2e[disagg_upload-gen_only-gb200_gpt-oss-120b-fp4_8k1k_con4_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-UCX] SKIP (https://nvbugs/5846166) -perf/test_perf_sanity.py::test_e2e[disagg_upload-gen_only-gb200_deepseek-r1-fp4_1k1k_con3072_ctx1_dep4_gen1_dep4_eplb0_mtp1_ccb-UCX] SKIP (https://nvbugs/5846166) -perf/test_perf_sanity.py::test_e2e[disagg_upload-gen_only-gb200_qwen3-235b-fp4_8k1k_con1024_ctx1_tp1_gen1_dep8_eplb0_mtp0_ccb-UCX] SKIP (https://nvbugs/5846166) -perf/test_perf_sanity.py::test_e2e[disagg_upload-gen_only-gb200_deepseek-r1-fp4_1k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX] SKIP (https://nvbugs/5846166) -perf/test_perf_sanity.py::test_e2e[disagg_upload-gen_only-gb200_deepseek-r1-fp4_8k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX] SKIP (https://nvbugs/5846166) -perf/test_perf_sanity.py::test_e2e[disagg_upload-gen_only-gb200_deepseek-r1-fp4_128k8k_con1_ctx1_pp8_gen1_tep8_eplb0_mtp3_ccb-UCX] SKIP (https://nvbugs/5846166) accuracy/test_disaggregated_serving.py::TestLlama3_1_8BInstruct::test_guided_decoding_with_eagle3[xgrammar-eagle3_one_model=True] SKIP (https://nvbugs/5879614) accuracy/test_disaggregated_serving.py::TestLlama3_1_8BInstruct::test_guided_decoding_with_eagle3[llguidance-eagle3_one_model=True] SKIP (https://nvbugs/5893116) accuracy/test_disaggregated_serving.py::TestLlama3_1_8BInstruct::test_ctx_pp_gen_tp_asymmetric[MMLU-gen_tp=2-ctx_pp=4] SKIP (https://nvbugs/5875522) @@ -384,6 +367,8 @@ full:RTXPro6000D/accuracy/test_disaggregated_serving.py::TestDeepSeekV3Lite::tes full:RTXPro6000D/accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_nvfp4_4gpus[moe_backend=CUTLASS-mtp_nextn=0-ep4-fp8kv=True-attention_dp=True-cuda_graph=True-overlap_scheduler=True-torch_compile=False] SKIP (https://nvbugs/5948435) full:RTXPro6000D/accuracy/test_disaggregated_serving.py::TestDeepSeekV3Lite::test_guided_decoding[llguidance-mtp_nextn=0] SKIP (https://nvbugs/5948428) accuracy/test_llm_api_pytorch.py::TestKimiK25::test_nvfp4[tp8] SKIP (https://nvbugs/5951789) +perf/test_perf_sanity.py::test_e2e[aggr_upload-deepseek_v32_fp4_grace_blackwell-v32_fp4_tep4_mtp3_1k1k] SKIP (https://nvbugspro.nvidia.com/bug/5919026) +perf/test_perf_sanity.py::test_e2e[aggr_upload-deepseek_v32_fp4_grace_blackwell-v32_fp4_tep4_mtp3_8k1k] SKIP (https://nvbugspro.nvidia.com/bug/5919026) accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_bfloat16[mtp_nextn=2-attention_dp=True-cuda_graph=True-overlap_scheduler=True-torch_compile=False-enable_chunked_prefill=False] SKIP (https://nvbugs/5955765) accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_no_kv_cache_reuse[quant_dtype=none-mtp_nextn=2-fp8kv=False-attention_dp=True-cuda_graph=True-overlap_scheduler=True] SKIP (https://nvbugs/5955773) accuracy/test_llm_api_pytorch.py::TestDeepSeekV32::test_fp8_blockscale[baseline_mtp1] SKIP (https://nvbugs/5955792) diff --git a/tests/scripts/perf-sanity/README.md b/tests/scripts/perf-sanity/README.md index 4cb9619855c6..ada4a8c1bc1b 100644 --- a/tests/scripts/perf-sanity/README.md +++ b/tests/scripts/perf-sanity/README.md @@ -32,13 +32,28 @@ The submit scripts generate `slurm_launch.sh` from draft templates: | `jenkins/scripts/perf/local/submit.py` | Aggregated (local) | `jenkins/scripts/perf/aggregated/slurm_launch_draft.sh` | | `jenkins/scripts/perf/local/submit.py` | Disaggregated (local) | `jenkins/scripts/perf/disaggregated/slurm_launch_draft.sh` | +## Environment Variables + +The config folder paths can be overridden via environment variables. Both submit scripts (`local/submit.py` and `disaggregated/submit.py`) propagate these into the pytest execution environment. + +| Variable | Default | Description | +|----------|---------|-------------| +| `AGG_CONFIG_FOLDER` | `tests/scripts/perf-sanity/aggregated` | Path to aggregated config YAML files | +| `DISAGG_CONFIG_FOLDER` | `tests/scripts/perf-sanity/disaggregated` | Path to disaggregated config YAML files | + +**Example**: Run with custom config folders: +```bash +AGG_CONFIG_FOLDER=my/custom/agg DISAGG_CONFIG_FOLDER=my/custom/disagg \ + python jenkins/scripts/perf/local/submit.py ... +``` + ## Configuration Files There are two modes for perf sanity tests: aggregated (aggr) and disaggregated (disagg). ### Aggregated Mode Config Files -**Location**: `tests/scripts/perf-sanity` +**Location**: `tests/scripts/perf-sanity/aggregated` **File Naming**: `xxx.yaml` where words are connected by `_` (underscore), not `-` (hyphen). @@ -52,7 +67,7 @@ There are two modes for perf sanity tests: aggregated (aggr) and disaggregated ( ### Disaggregated Mode Config Files -**Location**: `tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity` +**Location**: `tests/scripts/perf-sanity/disaggregated` **File Naming**: `xxx.yaml` (can contain `-` hyphen). @@ -66,7 +81,7 @@ In each test db yml file (with keyword `perf_sanity`), there are four test types ### 1. Normal Aggregated Test -Uses aggregated config files from `tests/scripts/perf-sanity`. +Uses aggregated config files from `tests/scripts/perf-sanity/aggregated`. **Format**: ``` @@ -199,8 +214,8 @@ When working with perf sanity tests, use these paths: | Resource | Path | |----------|------| | Pytest script | `tests/integration/defs/perf/test_perf_sanity.py` | -| Aggregated configs | `tests/scripts/perf-sanity/*.yaml` | -| Disaggregated configs | `tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/*.yaml` | +| Aggregated configs | `tests/scripts/perf-sanity/aggregated/*.yaml` | +| Disaggregated configs | `tests/scripts/perf-sanity/disaggregated/*.yaml` | | CI submit (disagg only) | `jenkins/scripts/perf/disaggregated/submit.py` | | Local submit (all) | `jenkins/scripts/perf/local/submit.py` | | Jenkins pipeline | `jenkins/L0_Test.groovy` | diff --git a/tests/scripts/perf-sanity/config_database_b200_nvl.yaml b/tests/scripts/perf-sanity/aggregated/config_database_b200_nvl.yaml similarity index 100% rename from tests/scripts/perf-sanity/config_database_b200_nvl.yaml rename to tests/scripts/perf-sanity/aggregated/config_database_b200_nvl.yaml diff --git a/tests/scripts/perf-sanity/config_database_h200_sxm.yaml b/tests/scripts/perf-sanity/aggregated/config_database_h200_sxm.yaml similarity index 100% rename from tests/scripts/perf-sanity/config_database_h200_sxm.yaml rename to tests/scripts/perf-sanity/aggregated/config_database_h200_sxm.yaml diff --git a/tests/scripts/perf-sanity/deepseek_r1_fp4_v2_2_nodes_grace_blackwell.yaml b/tests/scripts/perf-sanity/aggregated/deepseek_r1_fp4_v2_2_nodes_grace_blackwell.yaml similarity index 100% rename from tests/scripts/perf-sanity/deepseek_r1_fp4_v2_2_nodes_grace_blackwell.yaml rename to tests/scripts/perf-sanity/aggregated/deepseek_r1_fp4_v2_2_nodes_grace_blackwell.yaml diff --git a/tests/scripts/perf-sanity/deepseek_r1_fp4_v2_blackwell.yaml b/tests/scripts/perf-sanity/aggregated/deepseek_r1_fp4_v2_blackwell.yaml similarity index 100% rename from tests/scripts/perf-sanity/deepseek_r1_fp4_v2_blackwell.yaml rename to tests/scripts/perf-sanity/aggregated/deepseek_r1_fp4_v2_blackwell.yaml diff --git a/tests/scripts/perf-sanity/deepseek_r1_fp4_v2_grace_blackwell.yaml b/tests/scripts/perf-sanity/aggregated/deepseek_r1_fp4_v2_grace_blackwell.yaml similarity index 100% rename from tests/scripts/perf-sanity/deepseek_r1_fp4_v2_grace_blackwell.yaml rename to tests/scripts/perf-sanity/aggregated/deepseek_r1_fp4_v2_grace_blackwell.yaml diff --git a/tests/scripts/perf-sanity/deepseek_r1_fp8_blackwell.yaml b/tests/scripts/perf-sanity/aggregated/deepseek_r1_fp8_blackwell.yaml similarity index 100% rename from tests/scripts/perf-sanity/deepseek_r1_fp8_blackwell.yaml rename to tests/scripts/perf-sanity/aggregated/deepseek_r1_fp8_blackwell.yaml diff --git a/tests/scripts/perf-sanity/deepseek_v32_fp4_blackwell.yaml b/tests/scripts/perf-sanity/aggregated/deepseek_v32_fp4_blackwell.yaml similarity index 100% rename from tests/scripts/perf-sanity/deepseek_v32_fp4_blackwell.yaml rename to tests/scripts/perf-sanity/aggregated/deepseek_v32_fp4_blackwell.yaml diff --git a/tests/scripts/perf-sanity/deepseek_v32_fp4_grace_blackwell.yaml b/tests/scripts/perf-sanity/aggregated/deepseek_v32_fp4_grace_blackwell.yaml similarity index 100% rename from tests/scripts/perf-sanity/deepseek_v32_fp4_grace_blackwell.yaml rename to tests/scripts/perf-sanity/aggregated/deepseek_v32_fp4_grace_blackwell.yaml diff --git a/tests/scripts/perf-sanity/gb300_deepseek_r1_fp4_v2_2_nodes_grace_blackwell.yaml b/tests/scripts/perf-sanity/aggregated/gb300_deepseek_r1_fp4_v2_2_nodes_grace_blackwell.yaml similarity index 100% rename from tests/scripts/perf-sanity/gb300_deepseek_r1_fp4_v2_2_nodes_grace_blackwell.yaml rename to tests/scripts/perf-sanity/aggregated/gb300_deepseek_r1_fp4_v2_2_nodes_grace_blackwell.yaml diff --git a/tests/scripts/perf-sanity/gpt_oss_120b_fp4_blackwell.yaml b/tests/scripts/perf-sanity/aggregated/gpt_oss_120b_fp4_blackwell.yaml similarity index 100% rename from tests/scripts/perf-sanity/gpt_oss_120b_fp4_blackwell.yaml rename to tests/scripts/perf-sanity/aggregated/gpt_oss_120b_fp4_blackwell.yaml diff --git a/tests/scripts/perf-sanity/gpt_oss_120b_fp4_grace_blackwell.yaml b/tests/scripts/perf-sanity/aggregated/gpt_oss_120b_fp4_grace_blackwell.yaml similarity index 100% rename from tests/scripts/perf-sanity/gpt_oss_120b_fp4_grace_blackwell.yaml rename to tests/scripts/perf-sanity/aggregated/gpt_oss_120b_fp4_grace_blackwell.yaml diff --git a/tests/scripts/perf-sanity/k2_thinking_fp4_2_nodes_grace_blackwell.yaml b/tests/scripts/perf-sanity/aggregated/k2_thinking_fp4_2_nodes_grace_blackwell.yaml similarity index 100% rename from tests/scripts/perf-sanity/k2_thinking_fp4_2_nodes_grace_blackwell.yaml rename to tests/scripts/perf-sanity/aggregated/k2_thinking_fp4_2_nodes_grace_blackwell.yaml diff --git a/tests/scripts/perf-sanity/k2_thinking_fp4_blackwell.yaml b/tests/scripts/perf-sanity/aggregated/k2_thinking_fp4_blackwell.yaml similarity index 100% rename from tests/scripts/perf-sanity/k2_thinking_fp4_blackwell.yaml rename to tests/scripts/perf-sanity/aggregated/k2_thinking_fp4_blackwell.yaml diff --git a/tests/scripts/perf-sanity/k2_thinking_fp4_grace_blackwell.yaml b/tests/scripts/perf-sanity/aggregated/k2_thinking_fp4_grace_blackwell.yaml similarity index 100% rename from tests/scripts/perf-sanity/k2_thinking_fp4_grace_blackwell.yaml rename to tests/scripts/perf-sanity/aggregated/k2_thinking_fp4_grace_blackwell.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/b200_deepseek-r1-fp4_1k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/b200_deepseek-r1-fp4_1k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/b200_deepseek-r1-fp4_1k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/b200_deepseek-r1-fp4_1k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/b200_deepseek-r1-fp4_1k1k_con2048_ctx1_dep4_gen1_dep8_eplb0_mtp1_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/b200_deepseek-r1-fp4_1k1k_con2048_ctx1_dep4_gen1_dep8_eplb0_mtp1_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/b200_deepseek-r1-fp4_1k1k_con2048_ctx1_dep4_gen1_dep8_eplb0_mtp1_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/b200_deepseek-r1-fp4_1k1k_con2048_ctx1_dep4_gen1_dep8_eplb0_mtp1_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/b200_deepseek-r1-fp4_1k1k_con256_ctx1_dep4_gen1_dep8_eplb0_mtp3_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/b200_deepseek-r1-fp4_1k1k_con256_ctx1_dep4_gen1_dep8_eplb0_mtp3_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/b200_deepseek-r1-fp4_1k1k_con256_ctx1_dep4_gen1_dep8_eplb0_mtp3_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/b200_deepseek-r1-fp4_1k1k_con256_ctx1_dep4_gen1_dep8_eplb0_mtp3_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/b200_deepseek-r1-fp4_8k1k_con1536_ctx1_dep4_gen1_dep8_eplb0_mtp1_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/b200_deepseek-r1-fp4_8k1k_con1536_ctx1_dep4_gen1_dep8_eplb0_mtp1_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/b200_deepseek-r1-fp4_8k1k_con1536_ctx1_dep4_gen1_dep8_eplb0_mtp1_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/b200_deepseek-r1-fp4_8k1k_con1536_ctx1_dep4_gen1_dep8_eplb0_mtp1_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/b200_deepseek-r1-fp4_8k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/b200_deepseek-r1-fp4_8k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/b200_deepseek-r1-fp4_8k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/b200_deepseek-r1-fp4_8k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/b200_deepseek-r1-fp4_8k1k_con256_ctx1_dep4_gen1_dep8_eplb0_mtp1_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/b200_deepseek-r1-fp4_8k1k_con256_ctx1_dep4_gen1_dep8_eplb0_mtp1_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/b200_deepseek-r1-fp4_8k1k_con256_ctx1_dep4_gen1_dep8_eplb0_mtp1_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/b200_deepseek-r1-fp4_8k1k_con256_ctx1_dep4_gen1_dep8_eplb0_mtp1_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-r1-fp4_128k8k_con128_ctx1_pp8_gen1_dep16_eplb0_mtp2_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_deepseek-r1-fp4_128k8k_con128_ctx1_pp8_gen1_dep16_eplb0_mtp2_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-r1-fp4_128k8k_con128_ctx1_pp8_gen1_dep16_eplb0_mtp2_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_deepseek-r1-fp4_128k8k_con128_ctx1_pp8_gen1_dep16_eplb0_mtp2_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-r1-fp4_128k8k_con1_ctx1_pp8_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_deepseek-r1-fp4_128k8k_con1_ctx1_pp8_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-r1-fp4_128k8k_con1_ctx1_pp8_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_deepseek-r1-fp4_128k8k_con1_ctx1_pp8_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-r1-fp4_128k8k_con64_ctx1_pp8_gen1_dep32_eplb0_mtp3_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_deepseek-r1-fp4_128k8k_con64_ctx1_pp8_gen1_dep32_eplb0_mtp3_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-r1-fp4_128k8k_con64_ctx1_pp8_gen1_dep32_eplb0_mtp3_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_deepseek-r1-fp4_128k8k_con64_ctx1_pp8_gen1_dep32_eplb0_mtp3_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-r1-fp4_1k1k_con1024_ctx1_dep4_gen1_dep32_eplb0_mtp3_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_deepseek-r1-fp4_1k1k_con1024_ctx1_dep4_gen1_dep32_eplb0_mtp3_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-r1-fp4_1k1k_con1024_ctx1_dep4_gen1_dep32_eplb0_mtp3_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_deepseek-r1-fp4_1k1k_con1024_ctx1_dep4_gen1_dep32_eplb0_mtp3_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-r1-fp4_1k1k_con1024_ctx1_dep4_gen1_dep8_eplb0_mtp0_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_deepseek-r1-fp4_1k1k_con1024_ctx1_dep4_gen1_dep8_eplb0_mtp0_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-r1-fp4_1k1k_con1024_ctx1_dep4_gen1_dep8_eplb0_mtp0_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_deepseek-r1-fp4_1k1k_con1024_ctx1_dep4_gen1_dep8_eplb0_mtp0_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-r1-fp4_1k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_deepseek-r1-fp4_1k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-r1-fp4_1k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_deepseek-r1-fp4_1k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-r1-fp4_1k1k_con3072_ctx1_dep4_gen1_dep4_eplb0_mtp1_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_deepseek-r1-fp4_1k1k_con3072_ctx1_dep4_gen1_dep4_eplb0_mtp1_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-r1-fp4_1k1k_con3072_ctx1_dep4_gen1_dep4_eplb0_mtp1_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_deepseek-r1-fp4_1k1k_con3072_ctx1_dep4_gen1_dep4_eplb0_mtp1_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-r1-fp4_8k1k_con1024_ctx1_dep4_gen1_dep32_eplb0_mtp3_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_deepseek-r1-fp4_8k1k_con1024_ctx1_dep4_gen1_dep32_eplb0_mtp3_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-r1-fp4_8k1k_con1024_ctx1_dep4_gen1_dep32_eplb0_mtp3_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_deepseek-r1-fp4_8k1k_con1024_ctx1_dep4_gen1_dep32_eplb0_mtp3_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-r1-fp4_8k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_deepseek-r1-fp4_8k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-r1-fp4_8k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_deepseek-r1-fp4_8k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-r1-fp4_8k1k_con4096_ctx1_dep4_gen1_dep16_eplb0_mtp1_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_deepseek-r1-fp4_8k1k_con4096_ctx1_dep4_gen1_dep16_eplb0_mtp1_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-r1-fp4_8k1k_con4096_ctx1_dep4_gen1_dep16_eplb0_mtp1_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_deepseek-r1-fp4_8k1k_con4096_ctx1_dep4_gen1_dep16_eplb0_mtp1_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-v32-fp4_1k1k_con1024_ctx1_dep4_gen1_dep32_eplb256_mtp3_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_deepseek-v32-fp4_1k1k_con1024_ctx1_dep4_gen1_dep32_eplb256_mtp3_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-v32-fp4_1k1k_con1024_ctx1_dep4_gen1_dep32_eplb256_mtp3_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_deepseek-v32-fp4_1k1k_con1024_ctx1_dep4_gen1_dep32_eplb256_mtp3_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-v32-fp4_1k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_deepseek-v32-fp4_1k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-v32-fp4_1k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_deepseek-v32-fp4_1k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-v32-fp4_1k1k_con2048_ctx1_dep4_gen1_dep4_eplb0_mtp1_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_deepseek-v32-fp4_1k1k_con2048_ctx1_dep4_gen1_dep4_eplb0_mtp1_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-v32-fp4_1k1k_con2048_ctx1_dep4_gen1_dep4_eplb0_mtp1_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_deepseek-v32-fp4_1k1k_con2048_ctx1_dep4_gen1_dep4_eplb0_mtp1_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-v32-fp4_32k4k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_deepseek-v32-fp4_32k4k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-v32-fp4_32k4k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_deepseek-v32-fp4_32k4k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-v32-fp4_32k4k_con2048_ctx1_dep4_gen1_dep32_eplb288_mtp1_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_deepseek-v32-fp4_32k4k_con2048_ctx1_dep4_gen1_dep32_eplb288_mtp1_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-v32-fp4_32k4k_con2048_ctx1_dep4_gen1_dep32_eplb288_mtp1_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_deepseek-v32-fp4_32k4k_con2048_ctx1_dep4_gen1_dep32_eplb288_mtp1_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-v32-fp4_32k4k_con256_ctx1_dep4_gen1_dep32_eplb0_mtp3_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_deepseek-v32-fp4_32k4k_con256_ctx1_dep4_gen1_dep32_eplb0_mtp3_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-v32-fp4_32k4k_con256_ctx1_dep4_gen1_dep32_eplb0_mtp3_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_deepseek-v32-fp4_32k4k_con256_ctx1_dep4_gen1_dep32_eplb0_mtp3_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-v32-fp4_8k1k_con1024_ctx1_dep4_gen1_dep32_eplb256_mtp3_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_deepseek-v32-fp4_8k1k_con1024_ctx1_dep4_gen1_dep32_eplb256_mtp3_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-v32-fp4_8k1k_con1024_ctx1_dep4_gen1_dep32_eplb256_mtp3_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_deepseek-v32-fp4_8k1k_con1024_ctx1_dep4_gen1_dep32_eplb256_mtp3_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-v32-fp4_8k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_deepseek-v32-fp4_8k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-v32-fp4_8k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_deepseek-v32-fp4_8k1k_con1_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-v32-fp4_8k1k_con4096_ctx1_dep4_gen1_dep32_eplb256_mtp0_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_deepseek-v32-fp4_8k1k_con4096_ctx1_dep4_gen1_dep32_eplb256_mtp0_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_deepseek-v32-fp4_8k1k_con4096_ctx1_dep4_gen1_dep32_eplb256_mtp0_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_deepseek-v32-fp4_8k1k_con4096_ctx1_dep4_gen1_dep32_eplb256_mtp0_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_gpt-oss-120b-fp4_1k1k_con2048_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_1k1k_con2048_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_gpt-oss-120b-fp4_1k1k_con2048_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_1k1k_con2048_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_gpt-oss-120b-fp4_1k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_1k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_gpt-oss-120b-fp4_1k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_1k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_gpt-oss-120b-fp4_1k1k_con64_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_1k1k_con64_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_gpt-oss-120b-fp4_1k1k_con64_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_1k1k_con64_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_gpt-oss-120b-fp4_8k1k_con128_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con128_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_gpt-oss-120b-fp4_8k1k_con128_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con128_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_gpt-oss-120b-fp4_8k1k_con4_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con4_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_gpt-oss-120b-fp4_8k1k_con4_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con4_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_gpt-oss-120b-fp4_8k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_gpt-oss-120b-fp4_8k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_kimi-k2-thinking-fp4_1k1k_con2048_ctx1_dep4_gen1_dep32_eplb384_mtp0_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_kimi-k2-thinking-fp4_1k1k_con2048_ctx1_dep4_gen1_dep32_eplb384_mtp0_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_kimi-k2-thinking-fp4_1k1k_con2048_ctx1_dep4_gen1_dep32_eplb384_mtp0_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_kimi-k2-thinking-fp4_1k1k_con2048_ctx1_dep4_gen1_dep32_eplb384_mtp0_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_kimi-k2-thinking-fp4_1k1k_con4096_ctx1_dep4_gen1_dep8_eplb0_mtp0_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_kimi-k2-thinking-fp4_1k1k_con4096_ctx1_dep4_gen1_dep8_eplb0_mtp0_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_kimi-k2-thinking-fp4_1k1k_con4096_ctx1_dep4_gen1_dep8_eplb0_mtp0_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_kimi-k2-thinking-fp4_1k1k_con4096_ctx1_dep4_gen1_dep8_eplb0_mtp0_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_kimi-k2-thinking-fp4_1k1k_con4_ctx1_dep4_gen1_tep4_eplb0_mtp0_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_kimi-k2-thinking-fp4_1k1k_con4_ctx1_dep4_gen1_tep4_eplb0_mtp0_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_kimi-k2-thinking-fp4_1k1k_con4_ctx1_dep4_gen1_tep4_eplb0_mtp0_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_kimi-k2-thinking-fp4_1k1k_con4_ctx1_dep4_gen1_tep4_eplb0_mtp0_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_kimi-k2-thinking-fp4_8k1k_con1024_ctx1_dep4_gen1_dep32_eplb416_mtp3_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_kimi-k2-thinking-fp4_8k1k_con1024_ctx1_dep4_gen1_dep32_eplb416_mtp3_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_kimi-k2-thinking-fp4_8k1k_con1024_ctx1_dep4_gen1_dep32_eplb416_mtp3_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_kimi-k2-thinking-fp4_8k1k_con1024_ctx1_dep4_gen1_dep32_eplb416_mtp3_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_kimi-k2-thinking-fp4_8k1k_con4096_ctx1_dep4_gen1_dep16_eplb384_mtp0_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_kimi-k2-thinking-fp4_8k1k_con4096_ctx1_dep4_gen1_dep16_eplb384_mtp0_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_kimi-k2-thinking-fp4_8k1k_con4096_ctx1_dep4_gen1_dep16_eplb384_mtp0_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_kimi-k2-thinking-fp4_8k1k_con4096_ctx1_dep4_gen1_dep16_eplb384_mtp0_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_kimi-k2-thinking-fp4_8k1k_con4_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_kimi-k2-thinking-fp4_8k1k_con4_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_kimi-k2-thinking-fp4_8k1k_con4_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_kimi-k2-thinking-fp4_8k1k_con4_ctx1_dep4_gen1_tep8_eplb0_mtp3_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_qwen3-235b-fp4_8k1k_con1024_ctx1_tp1_gen1_dep8_eplb0_mtp0_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_qwen3-235b-fp4_8k1k_con1024_ctx1_tp1_gen1_dep8_eplb0_mtp0_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_qwen3-235b-fp4_8k1k_con1024_ctx1_tp1_gen1_dep8_eplb0_mtp0_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_qwen3-235b-fp4_8k1k_con1024_ctx1_tp1_gen1_dep8_eplb0_mtp0_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_qwen3-235b-fp4_8k1k_con64_ctx1_tp1_gen1_tep4_eplb0_mtp0_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_qwen3-235b-fp4_8k1k_con64_ctx1_tp1_gen1_tep4_eplb0_mtp0_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb200_qwen3-235b-fp4_8k1k_con64_ctx1_tp1_gen1_tep4_eplb0_mtp0_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb200_qwen3-235b-fp4_8k1k_con64_ctx1_tp1_gen1_tep4_eplb0_mtp0_ccb-UCX.yaml diff --git a/tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb300_deepseek-r1-fp4_1k1k_con1024_ctx1_dep4_gen1_dep32_eplb0_mtp3_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/gb300_deepseek-r1-fp4_1k1k_con1024_ctx1_dep4_gen1_dep32_eplb0_mtp3_ccb-UCX.yaml similarity index 100% rename from tests/integration/defs/perf/disagg/test_configs/disagg/perf-sanity/gb300_deepseek-r1-fp4_1k1k_con1024_ctx1_dep4_gen1_dep32_eplb0_mtp3_ccb-UCX.yaml rename to tests/scripts/perf-sanity/disaggregated/gb300_deepseek-r1-fp4_1k1k_con1024_ctx1_dep4_gen1_dep32_eplb0_mtp3_ccb-UCX.yaml diff --git a/tests/test_common/error_utils.py b/tests/test_common/error_utils.py index f456aa8e137a..9cb57aaf88f7 100644 --- a/tests/test_common/error_utils.py +++ b/tests/test_common/error_utils.py @@ -1,6 +1,17 @@ import os -ERROR_KEYWORDS = ["RuntimeError", "out of memory", "ValueError", "FileNotFoundError"] +ERROR_KEYWORDS = [ + "RuntimeError", + "out of memory", + "ValueError", + "FileNotFoundError", + "ConnectionRefusedError", + "ClientConnectorError", + "CancelledError", + "TimeoutError", + "PMI2_Init failed to initialize", + "OSError", +] SLURM_LOG_TAIL_LINES = 200 # Number of lines to print from slurm job logs ERROR_CONTEXT_LINES = 100 # Number of lines to print before and after error line From 2729faa81cd9650d39f5ac564009d2b4c112e882 Mon Sep 17 00:00:00 2001 From: Chenfei Zhang Date: Thu, 5 Mar 2026 21:16:15 -0800 Subject: [PATCH 02/12] update Signed-off-by: Chenfei Zhang --- tests/integration/defs/perf/test_perf_sanity.py | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/tests/integration/defs/perf/test_perf_sanity.py b/tests/integration/defs/perf/test_perf_sanity.py index cef22a01b749..58e316d7f2fc 100644 --- a/tests/integration/defs/perf/test_perf_sanity.py +++ b/tests/integration/defs/perf/test_perf_sanity.py @@ -906,13 +906,13 @@ def get_config_dir(benchmark_mode: Optional[str]) -> str: str: Absolute config directory path """ if benchmark_mode in ("e2e", "gen_only", "ctx_only"): - config_folder = DISAGG_CONFIG_FOLDER + config_dir = DISAGG_CONFIG_FOLDER else: - config_folder = AGG_CONFIG_FOLDER + config_dir = AGG_CONFIG_FOLDER # If relative path, join with llm root - if not os.path.isabs(config_folder): - config_folder = os.path.join(get_llm_root(), config_folder) - return config_folder + if not os.path.isabs(config_dir): + config_dir = os.path.join(get_llm_root(), config_dir) + return config_dir class PerfSanityTestConfig: From 742eb7b9c0ebd7fcada2352ca568505e7bcbaf54 Mon Sep 17 00:00:00 2001 From: FredricZ-2007 <226039983+fredricz-20070104@users.noreply.github.com> Date: Fri, 6 Mar 2026 03:11:39 +0000 Subject: [PATCH 03/12] add qa perf cases Signed-off-by: FredricZ-2007 <226039983+fredricz-20070104@users.noreply.github.com> --- .../test_lists/qa/llm_perf_multinode.txt | 98 ++++++++++++++ ...1_gen1_dep16_bs64_eplb0_mtp3_ccb-NIXL.yaml | 105 +++++++++++++++ ...x1_gen1_dep16_bs64_eplb0_mtp3_ccb-UCX.yaml | 105 +++++++++++++++ ...1_gen1_dep32_bs16_eplb0_mtp3_ccb-NIXL.yaml | 105 +++++++++++++++ ...x1_gen1_dep32_bs16_eplb0_mtp3_ccb-UCX.yaml | 105 +++++++++++++++ ...x1_gen4_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml | 100 +++++++++++++++ ...tx1_gen4_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml | 100 +++++++++++++++ ..._gen1_dep16_bs128_eplb0_mtp1_ccb-NIXL.yaml | 105 +++++++++++++++ ...2_gen1_dep16_bs128_eplb0_mtp1_ccb-UCX.yaml | 105 +++++++++++++++ ...x1_gen1_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml | 91 +++++++++++++ ...tx1_gen1_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml | 91 +++++++++++++ ...pp4_gen13_tep4_bs1_eplb0_mtp0-Default.yaml | 99 ++++++++++++++ ..._pp4_gen5_tep4_bs4_eplb0_mtp0-Default.yaml | 99 ++++++++++++++ ..._pp4_gen6_tep8_bs1_eplb0_mtp3-Default.yaml | 106 +++++++++++++++ ..._pp4_gen7_tep8_bs1_eplb0_mtp0-Default.yaml | 99 ++++++++++++++ ..._pp4_gen8_tep4_bs2_eplb0_mtp0-Default.yaml | 99 ++++++++++++++ ..._pp4_gen8_tep8_bs1_eplb0_mtp0-Default.yaml | 99 ++++++++++++++ ...pp8_gen11_tep4_bs2_eplb0_mtp0-Default.yaml | 99 ++++++++++++++ ...pp8_gen14_tep4_bs1_eplb0_mtp0-Default.yaml | 99 ++++++++++++++ ...pp8_gen1_dep16_bs1_eplb0_mtp3-Default.yaml | 103 +++++++++++++++ ..._pp8_gen1_dep8_bs4_eplb0_mtp2-Default.yaml | 103 +++++++++++++++ ..._pp8_gen1_tep8_bs1_eplb0_mtp0-Default.yaml | 99 ++++++++++++++ ..._pp8_gen1_tep8_bs1_eplb0_mtp3-Default.yaml | 104 +++++++++++++++ ..._pp8_gen1_tep8_bs2_eplb0_mtp3-Default.yaml | 104 +++++++++++++++ ..._pp8_gen5_tep8_bs2_eplb0_mtp3-Default.yaml | 103 +++++++++++++++ ..._pp8_gen7_tep4_bs2_eplb0_mtp2-Default.yaml | 103 +++++++++++++++ ..._pp8_gen7_tep8_bs1_eplb0_mtp0-Default.yaml | 99 ++++++++++++++ ..._pp8_gen8_tep4_bs4_eplb0_mtp0-Default.yaml | 99 ++++++++++++++ ..._pp4_gen7_tep8_bs2_eplb0_mtp3-Default.yaml | 105 +++++++++++++++ ...pp8_gen1_dep16_bs8_eplb0_mtp0-Default.yaml | 98 ++++++++++++++ ...pp8_gen1_dep32_bs2_eplb0_mtp0-Default.yaml | 98 ++++++++++++++ ...pp4_gen1_dep8_bs16_eplb0_mtp1-Default.yaml | 104 +++++++++++++++ ...p8_gen1_dep16_bs16_eplb0_mtp0-Default.yaml | 98 ++++++++++++++ ...pp8_gen1_dep16_bs8_eplb0_mtp2-Default.yaml | 103 +++++++++++++++ ...pp8_gen1_dep32_bs2_eplb0_mtp3-Default.yaml | 103 +++++++++++++++ ...pp8_gen1_dep32_bs4_eplb0_mtp0-Default.yaml | 98 ++++++++++++++ ...p4_gen1_dep16_bs16_eplb0_mtp0-Default.yaml | 98 ++++++++++++++ ...pp4_gen1_dep16_bs8_eplb0_mtp3-Default.yaml | 105 +++++++++++++++ ...pp4_gen1_dep32_bs2_eplb0_mtp3-Default.yaml | 105 +++++++++++++++ ...pp4_gen1_dep32_bs4_eplb0_mtp0-Default.yaml | 98 ++++++++++++++ ...p4_gen1_dep16_bs16_eplb0_mtp1-Default.yaml | 105 +++++++++++++++ ...p4_gen1_dep16_bs32_eplb0_mtp0-Default.yaml | 98 ++++++++++++++ ...p4_gen1_dep16_bs32_eplb0_mtp1-Default.yaml | 105 +++++++++++++++ ...pp4_gen1_dep32_bs4_eplb0_mtp3-Default.yaml | 105 +++++++++++++++ ...pp4_gen1_dep32_bs8_eplb0_mtp0-Default.yaml | 98 ++++++++++++++ ...pp4_gen1_dep32_bs8_eplb0_mtp3-Default.yaml | 105 +++++++++++++++ ...1_gen1_dep32_bs32_eplb0_mtp0_ccb-NIXL.yaml | 100 +++++++++++++++ ...x1_gen1_dep32_bs32_eplb0_mtp0_ccb-UCX.yaml | 99 ++++++++++++++ ...x1_gen4_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml | 100 +++++++++++++++ ...tx1_gen4_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml | 100 +++++++++++++++ ...x1_gen4_tep8_bs32_eplb0_mtp3_ccb-NIXL.yaml | 106 +++++++++++++++ ...tx1_gen4_tep8_bs32_eplb0_mtp3_ccb-UCX.yaml | 106 +++++++++++++++ ..._gen1_dep16_bs128_eplb0_mtp3_ccb-NIXL.yaml | 105 +++++++++++++++ ...2_gen1_dep16_bs128_eplb0_mtp3_ccb-UCX.yaml | 105 +++++++++++++++ ...1_gen1_dep32_bs128_eplb0_mtp3_ccb-UCX.yaml | 119 +++++++++++++++++ ...x1_gen3_tep8_bs16_eplb0_mtp3_ccb-NIXL.yaml | 106 +++++++++++++++ ...tx1_gen3_tep8_bs16_eplb0_mtp3_ccb-UCX.yaml | 106 +++++++++++++++ ...x1_gen3_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml | 100 +++++++++++++++ ...tx1_gen3_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml | 100 +++++++++++++++ ...6_gen1_dep16_bs64_eplb0_mtp0_ccb-NIXL.yaml | 99 ++++++++++++++ ...x6_gen1_dep16_bs64_eplb0_mtp0_ccb-UCX.yaml | 99 ++++++++++++++ ...8_gen1_dep32_bs16_eplb0_mtp3_ccb-NIXL.yaml | 105 +++++++++++++++ ...x8_gen1_dep32_bs16_eplb0_mtp3_ccb-UCX.yaml | 105 +++++++++++++++ ...gen1_dep16_bs64_eplb288_mtp3_ccb-NIXL.yaml | 113 ++++++++++++++++ ..._gen1_dep16_bs64_eplb288_mtp3_ccb-UCX.yaml | 113 ++++++++++++++++ ...gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml | 113 ++++++++++++++++ ..._gen1_dep32_bs16_eplb288_mtp3_ccb-UCX.yaml | 113 ++++++++++++++++ ...en1_dep16_bs128_eplb288_mtp1_ccb-NIXL.yaml | 113 ++++++++++++++++ ...gen1_dep16_bs128_eplb288_mtp1_ccb-UCX.yaml | 113 ++++++++++++++++ ...gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL.yaml | 106 +++++++++++++++ ...2_bs32_eplb288_mtp0_ccb-NIXL_kv-reuse.yaml | 106 +++++++++++++++ ..._gen1_dep32_bs32_eplb288_mtp0_ccb-UCX.yaml | 106 +++++++++++++++ ...en1_dep16_bs128_eplb288_mtp3_ccb-NIXL.yaml | 112 ++++++++++++++++ ...gen1_dep16_bs128_eplb288_mtp3_ccb-UCX.yaml | 112 ++++++++++++++++ ...1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT.yaml | 107 ++++++++++++++++ ..._dep32_bs128_eplb288_mtp3_ccb-DEFAULT.yaml | 113 ++++++++++++++++ ...gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL.yaml | 106 +++++++++++++++ ..._gen1_dep16_bs64_eplb288_mtp0_ccb-UCX.yaml | 106 +++++++++++++++ ...gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml | 112 ++++++++++++++++ ..._gen1_dep32_bs16_eplb288_mtp3_ccb-UCX.yaml | 112 ++++++++++++++++ ...gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL.yaml | 114 +++++++++++++++++ ...en1_dep16_bs128_eplb288_mtp3_ccb-NIXL.yaml | 120 +++++++++++++++++ ...1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT.yaml | 114 +++++++++++++++++ ..._dep32_bs128_eplb288_mtp3_ccb-DEFAULT.yaml | 121 ++++++++++++++++++ ...gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL.yaml | 114 +++++++++++++++++ ...gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml | 120 +++++++++++++++++ ...n1_dep32_bs1024_eplb384_mtp0_ccb-NIXL.yaml | 108 ++++++++++++++++ ...en1_dep32_bs256_eplb416_mtp0_ccb-NIXL.yaml | 108 ++++++++++++++++ 88 files changed, 9210 insertions(+) create mode 100644 tests/integration/test_lists/qa/llm_perf_multinode.txt create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen13_tep4_bs1_eplb0_mtp0-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen5_tep4_bs4_eplb0_mtp0-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen6_tep8_bs1_eplb0_mtp3-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen7_tep8_bs1_eplb0_mtp0-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep4_bs2_eplb0_mtp0-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep8_bs1_eplb0_mtp0-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen11_tep4_bs2_eplb0_mtp0-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen14_tep4_bs1_eplb0_mtp0-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep16_bs1_eplb0_mtp3-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep8_bs4_eplb0_mtp2-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp0-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp3-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs2_eplb0_mtp3-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen5_tep8_bs2_eplb0_mtp3-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep4_bs2_eplb0_mtp2-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep8_bs1_eplb0_mtp0-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen8_tep4_bs4_eplb0_mtp0-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp4_gen7_tep8_bs2_eplb0_mtp3-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep16_bs8_eplb0_mtp0-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep32_bs2_eplb0_mtp0-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp4_gen1_dep8_bs16_eplb0_mtp1-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs16_eplb0_mtp0-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs8_eplb0_mtp2-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs2_eplb0_mtp3-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs4_eplb0_mtp0-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs16_eplb0_mtp0-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs8_eplb0_mtp3-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs2_eplb0_mtp3-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs4_eplb0_mtp0-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs16_eplb0_mtp1-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs32_eplb0_mtp0-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep16_bs32_eplb0_mtp1-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs4_eplb0_mtp3-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp0-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp3-Default.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb0_mtp0_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb0_mtp0_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp3_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp3_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen1_dep32_bs128_eplb0_mtp3_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb0_mtp0_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb0_mtp0_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb0_mtp3_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb0_mtp3_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL_kv-reuse.yaml create mode 100644 tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT.yaml create mode 100644 tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT.yaml create mode 100644 tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT.yaml create mode 100644 tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT.yaml create mode 100644 tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_1k1k_ctx3_gen1_dep32_bs1024_eplb384_mtp0_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_8k1k_ctx8_gen1_dep32_bs256_eplb416_mtp0_ccb-NIXL.yaml diff --git a/tests/integration/test_lists/qa/llm_perf_multinode.txt b/tests/integration/test_lists/qa/llm_perf_multinode.txt new file mode 100644 index 000000000000..b2756599c5d1 --- /dev/null +++ b/tests/integration/test_lists/qa/llm_perf_multinode.txt @@ -0,0 +1,98 @@ +# disagg multi-node +# GB200 + GB300 supported cases +# perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_ccb-NIXL] +# perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_ccb-UCX] +# perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_ccb-NIXL] +# perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb0_mtp0_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb0_mtp0_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp3_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp3_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen1_dep32_bs128_eplb0_mtp3_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb0_mtp0_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb0_mtp0_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb0_mtp3_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb0_mtp3_ccb-UCX] + +# GB200 supported cases +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen11_tep4_bs2_eplb0_mtp0-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen14_tep4_bs1_eplb0_mtp0-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep16_bs1_eplb0_mtp3-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep8_bs4_eplb0_mtp2-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp0-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp3-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs2_eplb0_mtp3-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen5_tep8_bs2_eplb0_mtp3-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep4_bs2_eplb0_mtp2-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep8_bs1_eplb0_mtp0-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen8_tep4_bs4_eplb0_mtp0-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep16_bs8_eplb0_mtp0-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep32_bs2_eplb0_mtp0-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs16_eplb0_mtp0-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs8_eplb0_mtp2-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs2_eplb0_mtp3-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs4_eplb0_mtp0-Default] + +# GB300 supported cases +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp4_gen13_tep4_bs1_eplb0_mtp0-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp4_gen5_tep4_bs4_eplb0_mtp0-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp4_gen6_tep8_bs1_eplb0_mtp3-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp4_gen7_tep8_bs1_eplb0_mtp0-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep4_bs2_eplb0_mtp0-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep8_bs1_eplb0_mtp0-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx2_pp4_gen7_tep8_bs2_eplb0_mtp3-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx3_pp4_gen1_dep8_bs16_eplb0_mtp1-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs16_eplb0_mtp0-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs8_eplb0_mtp3-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs2_eplb0_mtp3-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs4_eplb0_mtp0-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs16_eplb0_mtp1-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs32_eplb0_mtp0-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep16_bs32_eplb0_mtp1-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs4_eplb0_mtp3-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp0-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp3-Default] + + + +# wideep multi-node +# GB200 + GB300 supported cases +perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL_kv-reuse] +perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT] +perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-r1-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT] +perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-v32-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT] +perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-v32-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT] +perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-v32-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-v32-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_kimi-k2-thinking-fp4_1k1k_ctx3_gen1_dep32_bs1024_eplb384_mtp0_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_kimi-k2-thinking-fp4_8k1k_ctx8_gen1_dep32_bs256_eplb416_mtp0_ccb-NIXL] +# GB200 supported cases +# GB300 supported cases diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_ccb-NIXL.yaml new file mode 100644 index 000000000000..2f94667a536c --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_ccb-NIXL.yaml @@ -0,0 +1,105 @@ +metadata: + model_name: Qwen3-235B-A22B-FP4 + precision: fp4 + model_dir_name: Qwen3-235B-A22B-FP4 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: 512 1024 + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 64 + max_num_tokens: 256 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: WIDEEP + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_ccb-UCX.yaml new file mode 100644 index 000000000000..fba25771bb18 --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_ccb-UCX.yaml @@ -0,0 +1,105 @@ +metadata: + model_name: Qwen3-235B-A22B-FP4 + precision: fp4 + model_dir_name: Qwen3-235B-A22B-FP4 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: 512 1024 + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 64 + max_num_tokens: 256 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: WIDEEP + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_ccb-NIXL.yaml new file mode 100644 index 000000000000..213c6ced5f33 --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_ccb-NIXL.yaml @@ -0,0 +1,105 @@ +metadata: + model_name: Qwen3-235B-A22B-FP4 + precision: fp4 + model_dir_name: Qwen3-235B-A22B-FP4 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '512' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 16 + max_num_tokens: 64 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.6 + dtype: fp8 + moe_config: + backend: WIDEEP + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_ccb-UCX.yaml new file mode 100644 index 000000000000..d7279d3b219a --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_ccb-UCX.yaml @@ -0,0 +1,105 @@ +metadata: + model_name: Qwen3-235B-A22B-FP4 + precision: fp4 + model_dir_name: Qwen3-235B-A22B-FP4 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '512' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 16 + max_num_tokens: 64 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.6 + dtype: fp8 + moe_config: + backend: WIDEEP + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml new file mode 100644 index 000000000000..1ff8177d2348 --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml @@ -0,0 +1,100 @@ +metadata: + model_name: Qwen3-235B-A22B-FP4 + precision: fp4 + model_dir_name: Qwen3-235B-A22B-FP4 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: 1 2 4 8 16 32 + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml new file mode 100644 index 000000000000..0437408d4317 --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml @@ -0,0 +1,100 @@ +metadata: + model_name: Qwen3-235B-A22B-FP4 + precision: fp4 + model_dir_name: Qwen3-235B-A22B-FP4 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: 1 2 4 8 16 32 + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_ccb-NIXL.yaml new file mode 100644 index 000000000000..eb9fd647b7dc --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_ccb-NIXL.yaml @@ -0,0 +1,105 @@ +metadata: + model_name: Qwen3-235B-A22B-FP4 + precision: fp4 + model_dir_name: Qwen3-235B-A22B-FP4 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '2048' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 2 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 128 + max_num_tokens: 256 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: WIDEEP + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_ccb-UCX.yaml new file mode 100644 index 000000000000..e468e51c7eac --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_ccb-UCX.yaml @@ -0,0 +1,105 @@ +metadata: + model_name: Qwen3-235B-A22B-FP4 + precision: fp4 + model_dir_name: Qwen3-235B-A22B-FP4 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '2048' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 2 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 128 + max_num_tokens: 256 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: WIDEEP + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml new file mode 100644 index 000000000000..7f4c66a24f90 --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml @@ -0,0 +1,91 @@ +# nvbugs: 5561153 +metadata: + model_name: Qwen3-235B-A22B-FP8 + precision: fp8 + model_dir_name: Qwen3-235B-A22B-FP8 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: 1 2 4 8 16 36 + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 64 + max_num_tokens: 2048 + max_seq_len: 2051 + cuda_graph_config: + enable_padding: true + max_batch_size: 128 + print_iter_log: true + kv_cache_config: + enable_block_reuse: true + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 2048 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + disable_overlap_scheduler: false + ctx: + max_batch_size: 32 + max_num_tokens: 2048 + max_seq_len: 2051 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: true + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 2048 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml new file mode 100644 index 000000000000..b6e6e3d82641 --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml @@ -0,0 +1,91 @@ +# nvbugs: 5561153 +metadata: + model_name: Qwen3-235B-A22B-FP8 + precision: fp8 + model_dir_name: Qwen3-235B-A22B-FP8 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: 1 2 4 8 16 36 + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 64 + max_num_tokens: 2048 + max_seq_len: 2051 + cuda_graph_config: + enable_padding: true + max_batch_size: 128 + print_iter_log: true + kv_cache_config: + enable_block_reuse: true + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 2048 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + disable_overlap_scheduler: false + ctx: + max_batch_size: 32 + max_num_tokens: 2048 + max_seq_len: 2051 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: true + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 2048 + backend: UCX diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen13_tep4_bs1_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen13_tep4_bs1_eplb0_mtp0-Default.yaml new file mode 100644 index 000000000000..a4ad607842d9 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen13_tep4_bs1_eplb0_mtp0-Default.yaml @@ -0,0 +1,99 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 13 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 1 + max_num_tokens: 128 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 4 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen5_tep4_bs4_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen5_tep4_bs4_eplb0_mtp0-Default.yaml new file mode 100644 index 000000000000..f2b1074ea4fc --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen5_tep4_bs4_eplb0_mtp0-Default.yaml @@ -0,0 +1,99 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '4' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 5 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 4 + max_num_tokens: 128 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 4 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen6_tep8_bs1_eplb0_mtp3-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen6_tep8_bs1_eplb0_mtp3-Default.yaml new file mode 100644 index 000000000000..3f9ef0ebaa12 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen6_tep8_bs1_eplb0_mtp3-Default.yaml @@ -0,0 +1,106 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 6 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 1 + # mtp_size=3 ⇒ max_num_tokens = 1 * (3 + 1) = 4 + max_num_tokens: 4 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 4 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen7_tep8_bs1_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen7_tep8_bs1_eplb0_mtp0-Default.yaml new file mode 100644 index 000000000000..ca20d690388a --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen7_tep8_bs1_eplb0_mtp0-Default.yaml @@ -0,0 +1,99 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 7 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 1 + max_num_tokens: 128 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 4 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep4_bs2_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep4_bs2_eplb0_mtp0-Default.yaml new file mode 100644 index 000000000000..a22245dee91a --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep4_bs2_eplb0_mtp0-Default.yaml @@ -0,0 +1,99 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '2' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 8 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 4 + max_batch_size: 2 + max_num_tokens: 128 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 4 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep8_bs1_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep8_bs1_eplb0_mtp0-Default.yaml new file mode 100644 index 000000000000..59d835780b16 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep8_bs1_eplb0_mtp0-Default.yaml @@ -0,0 +1,99 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 8 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 1 + max_num_tokens: 128 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 4 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen11_tep4_bs2_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen11_tep4_bs2_eplb0_mtp0-Default.yaml new file mode 100644 index 000000000000..5be853b4ac63 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen11_tep4_bs2_eplb0_mtp0-Default.yaml @@ -0,0 +1,99 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '2' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 11 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 2 + max_num_tokens: 128 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 8 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen14_tep4_bs1_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen14_tep4_bs1_eplb0_mtp0-Default.yaml new file mode 100644 index 000000000000..dfbf6222ed95 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen14_tep4_bs1_eplb0_mtp0-Default.yaml @@ -0,0 +1,99 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 14 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 1 + max_num_tokens: 128 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 8 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep16_bs1_eplb0_mtp3-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep16_bs1_eplb0_mtp3-Default.yaml new file mode 100644 index 000000000000..55972145a59f --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep16_bs1_eplb0_mtp3-Default.yaml @@ -0,0 +1,103 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 1 + # mtp_size=3 ⇒ max_num_tokens = 1 * (3 + 1) = 4 + max_num_tokens: 4 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: &id001 + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 8 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + speculative_config: *id001 + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep8_bs4_eplb0_mtp2-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep8_bs4_eplb0_mtp2-Default.yaml new file mode 100644 index 000000000000..02bc0863a9a0 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep8_bs4_eplb0_mtp2-Default.yaml @@ -0,0 +1,103 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '4' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 4 + # mtp_size=2 ⇒ max_num_tokens = 4 * (2 + 1) = 12 + max_num_tokens: 12 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: &id001 + decoding_type: MTP + num_nextn_predict_layers: 2 + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 8 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + speculative_config: *id001 + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp0-Default.yaml new file mode 100644 index 000000000000..fd3ad8c0f1be --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp0-Default.yaml @@ -0,0 +1,99 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 1 + max_num_tokens: 128 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 8 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp3-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp3-Default.yaml new file mode 100644 index 000000000000..1cd58fc3936e --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp3-Default.yaml @@ -0,0 +1,104 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 1 + # mtp_size=3 ⇒ max_num_tokens = 1 * (3 + 1) = 4 + max_num_tokens: 4 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + speculative_config: &id001 + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 8 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + speculative_config: *id001 + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs2_eplb0_mtp3-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs2_eplb0_mtp3-Default.yaml new file mode 100644 index 000000000000..1565c88347c7 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs2_eplb0_mtp3-Default.yaml @@ -0,0 +1,104 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '2' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 2 + # mtp_size=3 ⇒ max_num_tokens = 2 * (3 + 1) = 8 + max_num_tokens: 8 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + speculative_config: &id001 + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 8 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + speculative_config: *id001 + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen5_tep8_bs2_eplb0_mtp3-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen5_tep8_bs2_eplb0_mtp3-Default.yaml new file mode 100644 index 000000000000..281ab8215104 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen5_tep8_bs2_eplb0_mtp3-Default.yaml @@ -0,0 +1,103 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '2' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 5 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 2 + max_num_tokens: 8 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + speculative_config: &id001 + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 8 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + speculative_config: *id001 + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep4_bs2_eplb0_mtp2-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep4_bs2_eplb0_mtp2-Default.yaml new file mode 100644 index 000000000000..77e113fec290 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep4_bs2_eplb0_mtp2-Default.yaml @@ -0,0 +1,103 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '2' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 7 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 2 + max_num_tokens: 6 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + speculative_config: &id001 + decoding_type: MTP + num_nextn_predict_layers: 2 + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 8 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + speculative_config: *id001 + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep8_bs1_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep8_bs1_eplb0_mtp0-Default.yaml new file mode 100644 index 000000000000..517b5c61e778 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep8_bs1_eplb0_mtp0-Default.yaml @@ -0,0 +1,99 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 7 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 1 + max_num_tokens: 128 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 8 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen8_tep4_bs4_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen8_tep4_bs4_eplb0_mtp0-Default.yaml new file mode 100644 index 000000000000..449fd368a3a8 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen8_tep4_bs4_eplb0_mtp0-Default.yaml @@ -0,0 +1,99 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '4' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 8 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 4 + max_num_tokens: 128 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 8 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp4_gen7_tep8_bs2_eplb0_mtp3-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp4_gen7_tep8_bs2_eplb0_mtp3-Default.yaml new file mode 100644 index 000000000000..d794643060ab --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp4_gen7_tep8_bs2_eplb0_mtp3-Default.yaml @@ -0,0 +1,105 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '2' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 2 + num_gen_servers: 7 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 2 + max_num_tokens: 8 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 4 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep16_bs8_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep16_bs8_eplb0_mtp0-Default.yaml new file mode 100644 index 000000000000..ff9a9e62cf8b --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep16_bs8_eplb0_mtp0-Default.yaml @@ -0,0 +1,98 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '8' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 2 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 8 + max_num_tokens: 128 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 8 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep32_bs2_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep32_bs2_eplb0_mtp0-Default.yaml new file mode 100644 index 000000000000..d547dae70666 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep32_bs2_eplb0_mtp0-Default.yaml @@ -0,0 +1,98 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '2' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 2 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 2 + max_num_tokens: 128 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 8 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp4_gen1_dep8_bs16_eplb0_mtp1-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp4_gen1_dep8_bs16_eplb0_mtp1-Default.yaml new file mode 100644 index 000000000000..90d277005791 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp4_gen1_dep8_bs16_eplb0_mtp1-Default.yaml @@ -0,0 +1,104 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '128' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 3 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 16 + max_num_tokens: 32 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 4 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs16_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs16_eplb0_mtp0-Default.yaml new file mode 100644 index 000000000000..1c935ff7c416 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs16_eplb0_mtp0-Default.yaml @@ -0,0 +1,98 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '16' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 3 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 16 + max_num_tokens: 128 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 8 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs8_eplb0_mtp2-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs8_eplb0_mtp2-Default.yaml new file mode 100644 index 000000000000..ee9d98cdaa99 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs8_eplb0_mtp2-Default.yaml @@ -0,0 +1,103 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '8' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 3 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 8 + # mtp_size=2 ⇒ max_num_tokens = 8 * (2 + 1) = 24 + max_num_tokens: 24 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: &id001 + decoding_type: MTP + num_nextn_predict_layers: 2 + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 8 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + speculative_config: *id001 + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs2_eplb0_mtp3-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs2_eplb0_mtp3-Default.yaml new file mode 100644 index 000000000000..d69db0a1ca34 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs2_eplb0_mtp3-Default.yaml @@ -0,0 +1,103 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '2' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 3 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 2 + # mtp_size=3 ⇒ max_num_tokens = 2 * (3 + 1) = 8 + max_num_tokens: 8 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: &id001 + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 8 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + speculative_config: *id001 + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs4_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs4_eplb0_mtp0-Default.yaml new file mode 100644 index 000000000000..a51d1073e30c --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs4_eplb0_mtp0-Default.yaml @@ -0,0 +1,98 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '4' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 3 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 4 + max_num_tokens: 128 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 8 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs16_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs16_eplb0_mtp0-Default.yaml new file mode 100644 index 000000000000..05d6a10d3268 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs16_eplb0_mtp0-Default.yaml @@ -0,0 +1,98 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '256' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 5 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 16 + max_num_tokens: 128 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 4 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs8_eplb0_mtp3-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs8_eplb0_mtp3-Default.yaml new file mode 100644 index 000000000000..5befdee83318 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs8_eplb0_mtp3-Default.yaml @@ -0,0 +1,105 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '128' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 5 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 8 + # mtp_size=3 ⇒ max_num_tokens = 8 * (3 + 1) = 32 + max_num_tokens: 32 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 4 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs2_eplb0_mtp3-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs2_eplb0_mtp3-Default.yaml new file mode 100644 index 000000000000..e2bcac62240b --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs2_eplb0_mtp3-Default.yaml @@ -0,0 +1,105 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '64' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 5 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 2 + # mtp_size=3 ⇒ max_num_tokens = 2 * (3 + 1) = 8 + max_num_tokens: 8 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 4 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs4_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs4_eplb0_mtp0-Default.yaml new file mode 100644 index 000000000000..d449c173c778 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs4_eplb0_mtp0-Default.yaml @@ -0,0 +1,98 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '128' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 5 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 4 + max_num_tokens: 128 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 4 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs16_eplb0_mtp1-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs16_eplb0_mtp1-Default.yaml new file mode 100644 index 000000000000..90ed3bd0d33c --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs16_eplb0_mtp1-Default.yaml @@ -0,0 +1,105 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '256' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 7 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 16 + # mtp_size=1 ⇒ max_num_tokens = 16 * (1 + 1) = 32 + max_num_tokens: 32 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 4 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs32_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs32_eplb0_mtp0-Default.yaml new file mode 100644 index 000000000000..2eed9c9959c7 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs32_eplb0_mtp0-Default.yaml @@ -0,0 +1,98 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '512' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 7 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 4 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep16_bs32_eplb0_mtp1-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep16_bs32_eplb0_mtp1-Default.yaml new file mode 100644 index 000000000000..c9226167aa5a --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep16_bs32_eplb0_mtp1-Default.yaml @@ -0,0 +1,105 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '512' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 8 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 32 + # mtp_size=1 ⇒ max_num_tokens = 32 * (1 + 1) = 64 + max_num_tokens: 64 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 4 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs4_eplb0_mtp3-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs4_eplb0_mtp3-Default.yaml new file mode 100644 index 000000000000..e92e50d77f95 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs4_eplb0_mtp3-Default.yaml @@ -0,0 +1,105 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '128' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 8 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 4 + # mtp_size=3 ⇒ max_num_tokens = 4 * (3 + 1) = 16 + max_num_tokens: 16 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 4 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp0-Default.yaml new file mode 100644 index 000000000000..fb0c3d54835c --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp0-Default.yaml @@ -0,0 +1,98 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '256' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 8 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 8 + max_num_tokens: 128 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 4 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp3-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp3-Default.yaml new file mode 100644 index 000000000000..8740d2861c4f --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp3-Default.yaml @@ -0,0 +1,105 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 128k8k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '256' + input_length: 131072 + output_length: 8192 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 8 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 8 + # mtp_size=3 ⇒ max_num_tokens = 8 * (3 + 1) = 32 + max_num_tokens: 32 + max_seq_len: 139296 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + max_batch_size: 1 + max_num_tokens: 131104 + max_seq_len: 131104 + tensor_parallel_size: 1 + moe_expert_parallel_size: 1 + enable_attention_dp: false + pipeline_parallel_size: 4 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 131104 + backend: DEFAULT + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + moe_config: + backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb0_mtp0_ccb-NIXL.yaml new file mode 100644 index 000000000000..007c4b33258b --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb0_mtp0_ccb-NIXL.yaml @@ -0,0 +1,100 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1024' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + trtllm_wheel_path: '' + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 32 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: WIDEEP + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb0_mtp0_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb0_mtp0_ccb-UCX.yaml new file mode 100644 index 000000000000..d85ff79d08b9 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb0_mtp0_ccb-UCX.yaml @@ -0,0 +1,99 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1024' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 32 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: WIDEEP + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml new file mode 100644 index 000000000000..9f8ec3a9f956 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml @@ -0,0 +1,100 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: 1 2 4 8 16 32 + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml new file mode 100644 index 000000000000..6260fc5b8c0e --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml @@ -0,0 +1,100 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: 1 2 4 8 16 32 + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_ccb-NIXL.yaml new file mode 100644 index 000000000000..e3b9d07ff42b --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_ccb-NIXL.yaml @@ -0,0 +1,106 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: 1 2 4 8 16 32 + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_ccb-UCX.yaml new file mode 100644 index 000000000000..b483ce1c8c6a --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_ccb-UCX.yaml @@ -0,0 +1,106 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: 1 2 4 8 16 32 + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp3_ccb-NIXL.yaml new file mode 100644 index 000000000000..7bf0861937db --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp3_ccb-NIXL.yaml @@ -0,0 +1,105 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '2048' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 2 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 128 + max_num_tokens: 512 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: WIDEEP + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp3_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp3_ccb-UCX.yaml new file mode 100644 index 000000000000..9e6eda545937 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp3_ccb-UCX.yaml @@ -0,0 +1,105 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '2048' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 2 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 128 + max_num_tokens: 512 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: WIDEEP + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen1_dep32_bs128_eplb0_mtp3_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen1_dep32_bs128_eplb0_mtp3_ccb-UCX.yaml new file mode 100644 index 000000000000..c8f368acfccc --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen1_dep32_bs128_eplb0_mtp3_ccb-UCX.yaml @@ -0,0 +1,119 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k + config_index: -1 +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: gen_only + use_nv_sa_benchmark: true + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1024' + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + context_parallel_size: 1 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + pipeline_parallel_size: 1 + max_batch_size: 128 + max_num_tokens: 512 + max_seq_len: 9256 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 24 + - 32 + - 40 + - 48 + - 56 + - 64 + - 72 + - 80 + - 88 + - 96 + - 104 + - 112 + - 120 + - 128 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + moe_config: + backend: CUTEDSL + use_low_precision_moe_combine: true + load_balancer: + num_slots: 288 + layer_updates_per_iter: 1 + cache_transceiver_config: + max_tokens_in_buffer: 16384 + backend: UCX + stream_interval: 100 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + max_batch_size: 2 + max_num_tokens: 16896 + max_seq_len: 9256 + tensor_parallel_size: 4 + context_parallel_size: 1 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 16384 + backend: UCX + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_ccb-NIXL.yaml new file mode 100644 index 000000000000..6ff591400924 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_ccb-NIXL.yaml @@ -0,0 +1,106 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: 1 2 4 8 16 + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 3 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 16 + max_num_tokens: 64 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_ccb-UCX.yaml new file mode 100644 index 000000000000..7667c7903adc --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_ccb-UCX.yaml @@ -0,0 +1,106 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: 1 2 4 8 16 + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 3 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 16 + max_num_tokens: 64 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml new file mode 100644 index 000000000000..8a0bc5ca82ea --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml @@ -0,0 +1,100 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: 1 2 4 8 16 32 + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 3 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 32 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml new file mode 100644 index 000000000000..3328a559c3e3 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml @@ -0,0 +1,100 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: 1 2 4 8 16 32 + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 3 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 32 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb0_mtp0_ccb-NIXL.yaml new file mode 100644 index 000000000000..0b37895f1e62 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb0_mtp0_ccb-NIXL.yaml @@ -0,0 +1,99 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1024' + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 6 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 64 + max_num_tokens: 64 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: WIDEEP + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + ctx: + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb0_mtp0_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb0_mtp0_ccb-UCX.yaml new file mode 100644 index 000000000000..856de14f2f71 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb0_mtp0_ccb-UCX.yaml @@ -0,0 +1,99 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1024' + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 6 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 64 + max_num_tokens: 64 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: WIDEEP + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + ctx: + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb0_mtp3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb0_mtp3_ccb-NIXL.yaml new file mode 100644 index 000000000000..690afbff78be --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb0_mtp3_ccb-NIXL.yaml @@ -0,0 +1,105 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '512' + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 8 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 16 + max_num_tokens: 64 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: WIDEEP + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb0_mtp3_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb0_mtp3_ccb-UCX.yaml new file mode 100644 index 000000000000..115af8642dda --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb0_mtp3_ccb-UCX.yaml @@ -0,0 +1,105 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '512' + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 8 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 16 + max_num_tokens: 64 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: WIDEEP + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-NIXL.yaml new file mode 100644 index 000000000000..82ba1fc92c1d --- /dev/null +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-NIXL.yaml @@ -0,0 +1,113 @@ +metadata: + model_name: Qwen3-235B-A22B-FP4 + precision: fp4 + model_dir_name: Qwen3-235B-A22B-FP4 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k + dataset_file: disagg_datasets/deepseek-r1-1024-1024-100000-ratio-1_for_serve.json +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: gen_only + use_nv_sa_benchmark: false + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: 512 1024 + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + enable_layerwise_nvtx_marker: true + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + pipeline_parallel_size: 1 + max_batch_size: 64 + max_num_tokens: 256 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: WIDEEP + use_low_precision_moe_combine: true + load_balancer: + num_slots: 288 + layer_updates_per_iter: 1 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + enable_layerwise_nvtx_marker: true + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-UCX.yaml new file mode 100644 index 000000000000..431258ab8ebd --- /dev/null +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-UCX.yaml @@ -0,0 +1,113 @@ +metadata: + model_name: Qwen3-235B-A22B-FP4 + precision: fp4 + model_dir_name: Qwen3-235B-A22B-FP4 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k + dataset_file: disagg_datasets/deepseek-r1-1024-1024-100000-ratio-1_for_serve.json +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: gen_only + use_nv_sa_benchmark: false + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: 512 1024 + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + enable_layerwise_nvtx_marker: true + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + pipeline_parallel_size: 1 + max_batch_size: 64 + max_num_tokens: 256 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: WIDEEP + use_low_precision_moe_combine: true + load_balancer: + num_slots: 288 + layer_updates_per_iter: 1 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + enable_layerwise_nvtx_marker: true + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml new file mode 100644 index 000000000000..a1cb7b2a2430 --- /dev/null +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml @@ -0,0 +1,113 @@ +metadata: + model_name: Qwen3-235B-A22B-FP4 + precision: fp4 + model_dir_name: Qwen3-235B-A22B-FP4 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k + dataset_file: disagg_datasets/deepseek-r1-1024-1024-100000-ratio-1_for_serve.json +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: gen_only + use_nv_sa_benchmark: false + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '512' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + enable_layerwise_nvtx_marker: true + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + pipeline_parallel_size: 1 + max_batch_size: 16 + max_num_tokens: 64 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.6 + dtype: fp8 + moe_config: + backend: WIDEEP + use_low_precision_moe_combine: true + load_balancer: + num_slots: 288 + layer_updates_per_iter: 1 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + enable_layerwise_nvtx_marker: true + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_ccb-UCX.yaml new file mode 100644 index 000000000000..7711e130cae5 --- /dev/null +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_ccb-UCX.yaml @@ -0,0 +1,113 @@ +metadata: + model_name: Qwen3-235B-A22B-FP4 + precision: fp4 + model_dir_name: Qwen3-235B-A22B-FP4 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k + dataset_file: disagg_datasets/deepseek-r1-1024-1024-100000-ratio-1_for_serve.json +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: gen_only + use_nv_sa_benchmark: false + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '512' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + enable_layerwise_nvtx_marker: true + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + pipeline_parallel_size: 1 + max_batch_size: 16 + max_num_tokens: 64 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.6 + dtype: fp8 + moe_config: + backend: WIDEEP + use_low_precision_moe_combine: true + load_balancer: + num_slots: 288 + layer_updates_per_iter: 1 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + enable_layerwise_nvtx_marker: true + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_ccb-NIXL.yaml new file mode 100644 index 000000000000..873de8b2df3b --- /dev/null +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_ccb-NIXL.yaml @@ -0,0 +1,113 @@ +metadata: + model_name: Qwen3-235B-A22B-FP4 + precision: fp4 + model_dir_name: Qwen3-235B-A22B-FP4 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k + dataset_file: disagg_datasets/deepseek-r1-1024-1024-100000-ratio-1_for_serve.json +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: gen_only + use_nv_sa_benchmark: false + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '2048' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 2 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + enable_layerwise_nvtx_marker: true + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + pipeline_parallel_size: 1 + max_batch_size: 128 + max_num_tokens: 256 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: WIDEEP + use_low_precision_moe_combine: true + load_balancer: + num_slots: 288 + layer_updates_per_iter: 1 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 + ctx: + enable_layerwise_nvtx_marker: true + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_ccb-UCX.yaml new file mode 100644 index 000000000000..845b694cbbcd --- /dev/null +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_ccb-UCX.yaml @@ -0,0 +1,113 @@ +metadata: + model_name: Qwen3-235B-A22B-FP4 + precision: fp4 + model_dir_name: Qwen3-235B-A22B-FP4 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k + dataset_file: disagg_datasets/deepseek-r1-1024-1024-100000-ratio-1_for_serve.json +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: gen_only + use_nv_sa_benchmark: false + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '2048' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 2 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + enable_layerwise_nvtx_marker: true + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + pipeline_parallel_size: 1 + max_batch_size: 128 + max_num_tokens: 256 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: WIDEEP + use_low_precision_moe_combine: true + load_balancer: + num_slots: 288 + layer_updates_per_iter: 1 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 + ctx: + enable_layerwise_nvtx_marker: true + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL.yaml new file mode 100644 index 000000000000..93f3662b9f5f --- /dev/null +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL.yaml @@ -0,0 +1,106 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k + dataset_file: disagg_datasets/deepseek-r1-1024-1024-100000-ratio-1_for_serve.json +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: gen_only + use_nv_sa_benchmark: false + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1024' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + enable_layerwise_nvtx_marker: true + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 32 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: WIDEEP + load_balancer: + num_slots: 288 + layer_updates_per_iter: 1 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + ctx: + enable_layerwise_nvtx_marker: true + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL_kv-reuse.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL_kv-reuse.yaml new file mode 100644 index 000000000000..3aa96584e35a --- /dev/null +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL_kv-reuse.yaml @@ -0,0 +1,106 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k + dataset_file: disagg_datasets/deepseek-r1-1024-1024-100000-ratio-1_for_serve.json +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: gen_only + use_nv_sa_benchmark: false + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1024' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + enable_layerwise_nvtx_marker: true + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 32 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: true + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: WIDEEP + load_balancer: + num_slots: 288 + layer_updates_per_iter: 1 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + ctx: + enable_layerwise_nvtx_marker: true + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: true + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-UCX.yaml new file mode 100644 index 000000000000..10692cc27072 --- /dev/null +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-UCX.yaml @@ -0,0 +1,106 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k + dataset_file: disagg_datasets/deepseek-r1-1024-1024-100000-ratio-1_for_serve.json +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: gen_only + use_nv_sa_benchmark: false + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1024' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + enable_layerwise_nvtx_marker: true + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 32 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: WIDEEP + load_balancer: + num_slots: 288 + layer_updates_per_iter: 1 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + ctx: + enable_layerwise_nvtx_marker: true + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-NIXL.yaml new file mode 100644 index 000000000000..55a292499aaa --- /dev/null +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-NIXL.yaml @@ -0,0 +1,112 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k + dataset_file: disagg_datasets/deepseek-r1-1024-1024-100000-ratio-1_for_serve.json +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: gen_only + use_nv_sa_benchmark: false + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '2048' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 2 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + enable_layerwise_nvtx_marker: true + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + pipeline_parallel_size: 1 + max_batch_size: 128 + max_num_tokens: 512 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: WIDEEP + load_balancer: + num_slots: 288 + layer_updates_per_iter: 1 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + enable_layerwise_nvtx_marker: true + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-UCX.yaml new file mode 100644 index 000000000000..5c022fa2956a --- /dev/null +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-UCX.yaml @@ -0,0 +1,112 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k + dataset_file: disagg_datasets/deepseek-r1-1024-1024-100000-ratio-1_for_serve.json +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: gen_only + use_nv_sa_benchmark: false + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '2048' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 2 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + enable_layerwise_nvtx_marker: true + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + pipeline_parallel_size: 1 + max_batch_size: 128 + max_num_tokens: 512 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: WIDEEP + load_balancer: + num_slots: 288 + layer_updates_per_iter: 1 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + enable_layerwise_nvtx_marker: true + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT.yaml new file mode 100644 index 000000000000..ee04a0268d99 --- /dev/null +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT.yaml @@ -0,0 +1,107 @@ +# nvbugs: 5422621 +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k + config_index: 7 + dataset_file: disagg_datasets/deepseek-r1-1024-1024-100000-ratio-1_for_serve.json +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: gen_only + use_nv_sa_benchmark: false + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '12288' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 2 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + enable_layerwise_nvtx_marker: true + tensor_parallel_size: 48 + moe_expert_parallel_size: 48 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + pipeline_parallel_size: 1 + max_batch_size: 1024 + max_num_tokens: 1024 + max_seq_len: 2176 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: WIDEEP + load_balancer: + num_slots: 288 + layer_updates_per_iter: 1 + cache_transceiver_config: + max_tokens_in_buffer: 8320 + backend: DEFAULT + stream_interval: 20 + ctx: + enable_layerwise_nvtx_marker: true + max_batch_size: 4 + max_num_tokens: 4480 + max_seq_len: 2176 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8320 + backend: DEFAULT diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT.yaml new file mode 100644 index 000000000000..00c518c86481 --- /dev/null +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT.yaml @@ -0,0 +1,113 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k + dataset_file: disagg_datasets/deepseek-r1-8192-1024-200000-ratio-1_for_serve.json +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: disaggr-test + extra_args: "--gres=gpu:4" + numa_bind: true +hardware: + gpus_per_node: 4 + num_ctx_servers: 2 + num_gen_servers: 1 +benchmark: + mode: e2e + use_nv_sa_benchmark: false + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1024' + input_length: 8192 + output_length: 1024 + dataset_file: +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + enable_layerwise_nvtx_marker: true + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + pipeline_parallel_size: 1 + max_batch_size: 128 + max_num_tokens: 512 + max_seq_len: 9423 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + - 128 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.6 + dtype: fp8 + moe_config: + backend: WIDEEP + load_balancer: + num_slots: 288 + layer_updates_per_iter: 1 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + enable_layerwise_nvtx_marker: true + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9423 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: DEFAULT + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL.yaml new file mode 100644 index 000000000000..e6203c75c76b --- /dev/null +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL.yaml @@ -0,0 +1,106 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k + dataset_file: disagg_datasets/deepseek-r1-8192-1024-200000-ratio-1_for_serve.json +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: gen_only + use_nv_sa_benchmark: false + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1024' + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 6 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + enable_layerwise_nvtx_marker: true + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + pipeline_parallel_size: 1 + max_batch_size: 64 + max_num_tokens: 64 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: WIDEEP + load_balancer: + num_slots: 288 + layer_updates_per_iter: 1 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + ctx: + enable_layerwise_nvtx_marker: true + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-UCX.yaml new file mode 100644 index 000000000000..ff4c5276bf01 --- /dev/null +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-UCX.yaml @@ -0,0 +1,106 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k + dataset_file: disagg_datasets/deepseek-r1-8192-1024-200000-ratio-1_for_serve.json +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: gen_only + use_nv_sa_benchmark: false + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1024' + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 6 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + enable_layerwise_nvtx_marker: true + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + pipeline_parallel_size: 1 + max_batch_size: 64 + max_num_tokens: 64 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: WIDEEP + load_balancer: + num_slots: 288 + layer_updates_per_iter: 1 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + ctx: + enable_layerwise_nvtx_marker: true + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml new file mode 100644 index 000000000000..c9bc7351f8a8 --- /dev/null +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml @@ -0,0 +1,112 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k + dataset_file: disagg_datasets/deepseek-r1-8192-1024-200000-ratio-1_for_serve.json +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: gen_only + use_nv_sa_benchmark: false + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '512' + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 8 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + enable_layerwise_nvtx_marker: true + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + pipeline_parallel_size: 1 + max_batch_size: 16 + max_num_tokens: 64 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: WIDEEP + load_balancer: + num_slots: 288 + layer_updates_per_iter: 1 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + enable_layerwise_nvtx_marker: true + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-UCX.yaml new file mode 100644 index 000000000000..4185f89449fc --- /dev/null +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-UCX.yaml @@ -0,0 +1,112 @@ +metadata: + model_name: deepseek-r1-fp4 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k + dataset_file: disagg_datasets/deepseek-r1-8192-1024-200000-ratio-1_for_serve.json +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: gen_only + use_nv_sa_benchmark: false + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '512' + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 8 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + enable_layerwise_nvtx_marker: true + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + pipeline_parallel_size: 1 + max_batch_size: 16 + max_num_tokens: 64 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: WIDEEP + load_balancer: + num_slots: 288 + layer_updates_per_iter: 1 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + enable_layerwise_nvtx_marker: true + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL.yaml new file mode 100644 index 000000000000..0cbeb621611b --- /dev/null +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL.yaml @@ -0,0 +1,114 @@ +metadata: + model_name: deepseek-v32-fp4 + precision: fp4 + model_dir_name: DeepSeek-V3.2-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k + config_index: 1 + dataset_file: disagg_datasets/deepseek-v32-1024-1024-200000-ratio-1_for_serve.json +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 04:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: gen_only + use_nv_sa_benchmark: false + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1024' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false # Set to true to enable accuracy evaluation +worker_config: + gen: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 32 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: CUTEDSL + use_low_precision_moe_combine: true + load_balancer: + num_slots: 288 + layer_updates_per_iter: 1 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + moe_config: + backend: TRTLLM + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-NIXL.yaml new file mode 100644 index 000000000000..c3a4bee671cf --- /dev/null +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-NIXL.yaml @@ -0,0 +1,120 @@ +metadata: + model_name: deepseek-v32-fp4 + precision: fp4 + model_dir_name: DeepSeek-V3.2-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k + config_index: 0 + dataset_file: disagg_datasets/deepseek-v32-1024-1024-200000-ratio-1_for_serve.json +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 04:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: gen_only + use_nv_sa_benchmark: false + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '2048' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 2 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + pipeline_parallel_size: 1 + max_batch_size: 128 + max_num_tokens: 512 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: CUTEDSL + use_low_precision_moe_combine: true + load_balancer: + num_slots: 288 + layer_updates_per_iter: 1 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + moe_config: + backend: TRTLLM + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT.yaml new file mode 100644 index 000000000000..f3dcda394eb9 --- /dev/null +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT.yaml @@ -0,0 +1,114 @@ +# nvbugs: 5422621 +metadata: + model_name: deepseek-v32-fp4 + precision: fp4 + model_dir_name: DeepSeek-V3.2-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k + config_index: 7 + dataset_file: disagg_datasets/deepseek-v32-1024-1024-200000-ratio-1_for_serve.json +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 04:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: gen_only + use_nv_sa_benchmark: false + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '12288' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 2 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 48 + moe_expert_parallel_size: 48 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + pipeline_parallel_size: 1 + max_batch_size: 1024 + max_num_tokens: 1024 + max_seq_len: 2176 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: CUTEDSL + use_low_precision_moe_combine: true + load_balancer: + num_slots: 288 + layer_updates_per_iter: 1 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + cache_transceiver_config: + max_tokens_in_buffer: 8320 + backend: DEFAULT + stream_interval: 20 + ctx: + max_batch_size: 4 + max_num_tokens: 4480 + max_seq_len: 2176 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + moe_config: + backend: TRTLLM + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8320 + backend: DEFAULT diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT.yaml new file mode 100644 index 000000000000..8752c59e1f79 --- /dev/null +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT.yaml @@ -0,0 +1,121 @@ +metadata: + model_name: deepseek-v32-fp4 + precision: fp4 + model_dir_name: DeepSeek-V3.2-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k + config_index: 14 + dataset_file: disagg_datasets/deepseek-v32-8192-1024-200000-ratio-1_for_serve.json +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 04:00:00 + job_name: disaggr-test + extra_args: "--gres=gpu:4" + numa_bind: true +hardware: + gpus_per_node: 4 + num_ctx_servers: 2 + num_gen_servers: 1 +benchmark: + mode: e2e + use_nv_sa_benchmark: false + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1024' + input_length: 8192 + output_length: 1024 + dataset_file: +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + pipeline_parallel_size: 1 + max_batch_size: 128 + max_num_tokens: 512 + max_seq_len: 9423 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + - 128 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.6 + dtype: fp8 + moe_config: + backend: CUTEDSL + use_low_precision_moe_combine: true + load_balancer: + num_slots: 288 + layer_updates_per_iter: 1 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: DEFAULT + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9423 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: DEFAULT + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL.yaml new file mode 100644 index 000000000000..edec6340a751 --- /dev/null +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL.yaml @@ -0,0 +1,114 @@ +metadata: + model_name: deepseek-v32-fp4 + precision: fp4 + model_dir_name: DeepSeek-V3.2-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k + config_index: 5 + dataset_file: disagg_datasets/deepseek-v32-8192-1024-200000-ratio-1_for_serve.json +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 04:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: gen_only + use_nv_sa_benchmark: false + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1024' + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 6 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + pipeline_parallel_size: 1 + max_batch_size: 64 + max_num_tokens: 64 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: CUTEDSL + use_low_precision_moe_combine: true + load_balancer: + num_slots: 288 + layer_updates_per_iter: 1 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + ctx: + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml new file mode 100644 index 000000000000..05dc30cf54d3 --- /dev/null +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml @@ -0,0 +1,120 @@ +metadata: + model_name: deepseek-v32-fp4 + precision: fp4 + model_dir_name: DeepSeek-V3.2-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k + config_index: 4 + dataset_file: disagg_datasets/deepseek-v32-8192-1024-200000-ratio-1_for_serve.json +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 04:00:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: gen_only + use_nv_sa_benchmark: false + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '512' + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 8 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" + server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + pipeline_parallel_size: 1 + max_batch_size: 16 + max_num_tokens: 64 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: CUTEDSL + use_low_precision_moe_combine: true + load_balancer: + num_slots: 288 + layer_updates_per_iter: 1 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_1k1k_ctx3_gen1_dep32_bs1024_eplb384_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_1k1k_ctx3_gen1_dep32_bs1024_eplb384_mtp0_ccb-NIXL.yaml new file mode 100644 index 000000000000..44a756fc8df2 --- /dev/null +++ b/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_1k1k_ctx3_gen1_dep32_bs1024_eplb384_mtp0_ccb-NIXL.yaml @@ -0,0 +1,108 @@ +metadata: + model_name: kimi-k2-thinking-fp4 + precision: fp4 + model_dir_name: Kimi-K2-Thinking-NVFP4 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k + dataset_file: disagg_datasets/kimi-k2-1024-1024-20000-ratio-1_for_serve.json +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 00:45:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: gen_only + use_nv_sa_benchmark: false + multi_round: 8 + benchmark_ratio: 1.0 + streaming: true + concurrency_list: '16384' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 3 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + enable_layerwise_nvtx_marker: true + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + pipeline_parallel_size: 1 + max_batch_size: 1024 + max_num_tokens: 1024 + max_seq_len: 2068 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + - 1024 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.8 + dtype: fp8 + moe_config: + backend: WIDEEP + use_low_precision_moe_combine: true + load_balancer: + num_slots: 384 + layer_updates_per_iter: 1 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL + stream_interval: 100 + num_postprocess_workers: 4 + trust_remote_code: true + ctx: + enable_layerwise_nvtx_marker: true + max_batch_size: 8 + max_num_tokens: 8448 + max_seq_len: 1044 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL + trust_remote_code: true diff --git a/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_8k1k_ctx8_gen1_dep32_bs256_eplb416_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_8k1k_ctx8_gen1_dep32_bs256_eplb416_mtp0_ccb-NIXL.yaml new file mode 100644 index 000000000000..2584fd7908ef --- /dev/null +++ b/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_8k1k_ctx8_gen1_dep32_bs256_eplb416_mtp0_ccb-NIXL.yaml @@ -0,0 +1,108 @@ +metadata: + model_name: kimi-k2-thinking-fp4 + precision: fp4 + model_dir_name: Kimi-K2-Thinking-NVFP4 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k + dataset_file: disagg_datasets/kimi-k2-8192-1024-20000-ratio-1_for_serve.json +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 00:45:00 + job_name: unified-benchmark + extra_args: "--gres=gpu:4" + numa_bind: true +benchmark: + mode: gen_only + use_nv_sa_benchmark: false + multi_round: 8 + benchmark_ratio: 1.0 + streaming: true + concurrency_list: '8192' + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 8 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + enable_layerwise_nvtx_marker: true + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + pipeline_parallel_size: 1 + max_batch_size: 256 + max_num_tokens: 256 + max_seq_len: 9256 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + - 256 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.6 + dtype: fp8 + moe_config: + backend: WIDEEP + use_low_precision_moe_combine: true + load_balancer: + num_slots: 416 + layer_updates_per_iter: 1 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL + stream_interval: 100 + num_postprocess_workers: 4 + trust_remote_code: true + ctx: + enable_layerwise_nvtx_marker: true + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 8232 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL + trust_remote_code: true From 2eccf1ca7e4696d35b847733c15b9af318ded4b5 Mon Sep 17 00:00:00 2001 From: FredricZ-2007 <226039983+fredricz-20070104@users.noreply.github.com> Date: Fri, 6 Mar 2026 05:14:02 +0000 Subject: [PATCH 04/12] add disagg and wideep perf here Signed-off-by: FredricZ-2007 <226039983+fredricz-20070104@users.noreply.github.com> --- .../test_lists/qa/llm_perf_multinode.txt | 50 +++++++++---------- ...gen1_dep16_bs64_eplb288_mtp3_ccb-NIXL.yaml | 2 +- ..._gen1_dep16_bs64_eplb288_mtp3_ccb-UCX.yaml | 2 +- ...gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml | 2 +- ..._gen1_dep32_bs16_eplb288_mtp3_ccb-UCX.yaml | 2 +- ...en1_dep16_bs128_eplb288_mtp1_ccb-NIXL.yaml | 2 +- ...gen1_dep16_bs128_eplb288_mtp1_ccb-UCX.yaml | 2 +- ...gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL.yaml | 2 +- ...2_bs32_eplb288_mtp0_ccb-NIXL_kv-reuse.yaml | 2 +- ..._gen1_dep32_bs32_eplb288_mtp0_ccb-UCX.yaml | 2 +- ...en1_dep16_bs128_eplb288_mtp3_ccb-NIXL.yaml | 2 +- ...gen1_dep16_bs128_eplb288_mtp3_ccb-UCX.yaml | 2 +- ...1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT.yaml | 2 +- ...gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL.yaml | 2 +- ..._gen1_dep16_bs64_eplb288_mtp0_ccb-UCX.yaml | 2 +- ...gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml | 2 +- ..._gen1_dep32_bs16_eplb288_mtp3_ccb-UCX.yaml | 2 +- ...gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL.yaml | 2 +- ...en1_dep16_bs128_eplb288_mtp3_ccb-NIXL.yaml | 2 +- ...1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT.yaml | 2 +- ...gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL.yaml | 2 +- ...gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml | 2 +- ...n1_dep32_bs1024_eplb384_mtp0_ccb-NIXL.yaml | 2 +- ...en1_dep32_bs256_eplb416_mtp0_ccb-NIXL.yaml | 2 +- 24 files changed, 48 insertions(+), 48 deletions(-) diff --git a/tests/integration/test_lists/qa/llm_perf_multinode.txt b/tests/integration/test_lists/qa/llm_perf_multinode.txt index b2756599c5d1..5d048aa20495 100644 --- a/tests/integration/test_lists/qa/llm_perf_multinode.txt +++ b/tests/integration/test_lists/qa/llm_perf_multinode.txt @@ -69,30 +69,30 @@ perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx8_pp4_ge # wideep multi-node # GB200 + GB300 supported cases -perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-UCX] -perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_ccb-UCX] -perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_ccb-UCX] -perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL_kv-reuse] -perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-UCX] -perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-UCX] -perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT] -perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-r1-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT] -perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-UCX] -perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-UCX] -perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-v32-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT] -perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-v32-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT] -perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-v32-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_deepseek-v32-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_kimi-k2-thinking-fp4_1k1k_ctx3_gen1_dep32_bs1024_eplb384_mtp0_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[wideep-e2e-wideep_kimi-k2-thinking-fp4_8k1k_ctx8_gen1_dep32_bs256_eplb416_mtp0_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL_kv-reuse] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-wideep_deepseek-r1-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-v32-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-wideep_deepseek-v32-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-v32-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-v32-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_kimi-k2-thinking-fp4_1k1k_ctx3_gen1_dep32_bs1024_eplb384_mtp0_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_kimi-k2-thinking-fp4_8k1k_ctx8_gen1_dep32_bs256_eplb416_mtp0_ccb-NIXL] # GB200 supported cases # GB300 supported cases diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-NIXL.yaml index 82ba1fc92c1d..bcbf7532af69 100644 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-NIXL.yaml @@ -19,7 +19,7 @@ slurm: benchmark: mode: gen_only use_nv_sa_benchmark: false - multi_round: 8 + multi_round: 1 benchmark_ratio: 0.8 streaming: true concurrency_list: 512 1024 diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-UCX.yaml index 431258ab8ebd..579ed9445bec 100644 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-UCX.yaml @@ -19,7 +19,7 @@ slurm: benchmark: mode: gen_only use_nv_sa_benchmark: false - multi_round: 8 + multi_round: 1 benchmark_ratio: 0.8 streaming: true concurrency_list: 512 1024 diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml index a1cb7b2a2430..02a018e7505a 100644 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml @@ -19,7 +19,7 @@ slurm: benchmark: mode: gen_only use_nv_sa_benchmark: false - multi_round: 8 + multi_round: 1 benchmark_ratio: 0.8 streaming: true concurrency_list: '512' diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_ccb-UCX.yaml index 7711e130cae5..7ba14aabf6e8 100644 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_ccb-UCX.yaml @@ -19,7 +19,7 @@ slurm: benchmark: mode: gen_only use_nv_sa_benchmark: false - multi_round: 8 + multi_round: 1 benchmark_ratio: 0.8 streaming: true concurrency_list: '512' diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_ccb-NIXL.yaml index 873de8b2df3b..b32724a93b2b 100644 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_ccb-NIXL.yaml @@ -19,7 +19,7 @@ slurm: benchmark: mode: gen_only use_nv_sa_benchmark: false - multi_round: 8 + multi_round: 1 benchmark_ratio: 0.8 streaming: true concurrency_list: '2048' diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_ccb-UCX.yaml index 845b694cbbcd..142602f557ca 100644 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_ccb-UCX.yaml @@ -19,7 +19,7 @@ slurm: benchmark: mode: gen_only use_nv_sa_benchmark: false - multi_round: 8 + multi_round: 1 benchmark_ratio: 0.8 streaming: true concurrency_list: '2048' diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL.yaml index 93f3662b9f5f..b607270bf5d3 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL.yaml @@ -19,7 +19,7 @@ slurm: benchmark: mode: gen_only use_nv_sa_benchmark: false - multi_round: 8 + multi_round: 1 benchmark_ratio: 0.8 streaming: true concurrency_list: '1024' diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL_kv-reuse.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL_kv-reuse.yaml index 3aa96584e35a..da17daeeb042 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL_kv-reuse.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL_kv-reuse.yaml @@ -19,7 +19,7 @@ slurm: benchmark: mode: gen_only use_nv_sa_benchmark: false - multi_round: 8 + multi_round: 1 benchmark_ratio: 0.8 streaming: true concurrency_list: '1024' diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-UCX.yaml index 10692cc27072..b37f61f92a6f 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-UCX.yaml @@ -19,7 +19,7 @@ slurm: benchmark: mode: gen_only use_nv_sa_benchmark: false - multi_round: 8 + multi_round: 1 benchmark_ratio: 0.8 streaming: true concurrency_list: '1024' diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-NIXL.yaml index 55a292499aaa..d29762488d0b 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-NIXL.yaml @@ -19,7 +19,7 @@ slurm: benchmark: mode: gen_only use_nv_sa_benchmark: false - multi_round: 8 + multi_round: 1 benchmark_ratio: 0.8 streaming: true concurrency_list: '2048' diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-UCX.yaml index 5c022fa2956a..c5377581b2b7 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-UCX.yaml @@ -19,7 +19,7 @@ slurm: benchmark: mode: gen_only use_nv_sa_benchmark: false - multi_round: 8 + multi_round: 1 benchmark_ratio: 0.8 streaming: true concurrency_list: '2048' diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT.yaml index ee04a0268d99..4d42180625c0 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT.yaml @@ -21,7 +21,7 @@ slurm: benchmark: mode: gen_only use_nv_sa_benchmark: false - multi_round: 8 + multi_round: 1 benchmark_ratio: 0.8 streaming: true concurrency_list: '12288' diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL.yaml index e6203c75c76b..ee4256d4c528 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL.yaml @@ -19,7 +19,7 @@ slurm: benchmark: mode: gen_only use_nv_sa_benchmark: false - multi_round: 8 + multi_round: 1 benchmark_ratio: 0.8 streaming: true concurrency_list: '1024' diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-UCX.yaml index ff4c5276bf01..38cd607e96cd 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-UCX.yaml @@ -19,7 +19,7 @@ slurm: benchmark: mode: gen_only use_nv_sa_benchmark: false - multi_round: 8 + multi_round: 1 benchmark_ratio: 0.8 streaming: true concurrency_list: '1024' diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml index c9bc7351f8a8..2fa89f9fff9e 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml @@ -19,7 +19,7 @@ slurm: benchmark: mode: gen_only use_nv_sa_benchmark: false - multi_round: 8 + multi_round: 1 benchmark_ratio: 0.8 streaming: true concurrency_list: '512' diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-UCX.yaml index 4185f89449fc..a9ee6c3ed1ae 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-UCX.yaml @@ -19,7 +19,7 @@ slurm: benchmark: mode: gen_only use_nv_sa_benchmark: false - multi_round: 8 + multi_round: 1 benchmark_ratio: 0.8 streaming: true concurrency_list: '512' diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL.yaml index 0cbeb621611b..bf9bdc3c113d 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL.yaml @@ -20,7 +20,7 @@ slurm: benchmark: mode: gen_only use_nv_sa_benchmark: false - multi_round: 8 + multi_round: 1 benchmark_ratio: 0.8 streaming: true concurrency_list: '1024' diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-NIXL.yaml index c3a4bee671cf..e65b3e2fa858 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-NIXL.yaml @@ -20,7 +20,7 @@ slurm: benchmark: mode: gen_only use_nv_sa_benchmark: false - multi_round: 8 + multi_round: 1 benchmark_ratio: 0.8 streaming: true concurrency_list: '2048' diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT.yaml index f3dcda394eb9..6302846bb6b1 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT.yaml @@ -21,7 +21,7 @@ slurm: benchmark: mode: gen_only use_nv_sa_benchmark: false - multi_round: 8 + multi_round: 1 benchmark_ratio: 0.8 streaming: true concurrency_list: '12288' diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL.yaml index edec6340a751..a1f41602e9d1 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL.yaml @@ -20,7 +20,7 @@ slurm: benchmark: mode: gen_only use_nv_sa_benchmark: false - multi_round: 8 + multi_round: 1 benchmark_ratio: 0.8 streaming: true concurrency_list: '1024' diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml index 05dc30cf54d3..6219af7ebc61 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml @@ -20,7 +20,7 @@ slurm: benchmark: mode: gen_only use_nv_sa_benchmark: false - multi_round: 8 + multi_round: 1 benchmark_ratio: 0.8 streaming: true concurrency_list: '512' diff --git a/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_1k1k_ctx3_gen1_dep32_bs1024_eplb384_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_1k1k_ctx3_gen1_dep32_bs1024_eplb384_mtp0_ccb-NIXL.yaml index 44a756fc8df2..aa99610ab548 100644 --- a/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_1k1k_ctx3_gen1_dep32_bs1024_eplb384_mtp0_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_1k1k_ctx3_gen1_dep32_bs1024_eplb384_mtp0_ccb-NIXL.yaml @@ -19,7 +19,7 @@ slurm: benchmark: mode: gen_only use_nv_sa_benchmark: false - multi_round: 8 + multi_round: 1 benchmark_ratio: 1.0 streaming: true concurrency_list: '16384' diff --git a/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_8k1k_ctx8_gen1_dep32_bs256_eplb416_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_8k1k_ctx8_gen1_dep32_bs256_eplb416_mtp0_ccb-NIXL.yaml index 2584fd7908ef..7c37876ced04 100644 --- a/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_8k1k_ctx8_gen1_dep32_bs256_eplb416_mtp0_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_8k1k_ctx8_gen1_dep32_bs256_eplb416_mtp0_ccb-NIXL.yaml @@ -19,7 +19,7 @@ slurm: benchmark: mode: gen_only use_nv_sa_benchmark: false - multi_round: 8 + multi_round: 1 benchmark_ratio: 1.0 streaming: true concurrency_list: '8192' From a314b24f910b7d1607b3e284bcd043906c7f643d Mon Sep 17 00:00:00 2001 From: FredricZ-2007 <226039983+fredricz-20070104@users.noreply.github.com> Date: Fri, 6 Mar 2026 05:53:44 +0000 Subject: [PATCH 05/12] fx error configs Signed-off-by: FredricZ-2007 <226039983+fredricz-20070104@users.noreply.github.com> --- ...fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT.yaml | 2 +- ...fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT.yaml | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT.yaml index 00c518c86481..42ab879197dd 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT.yaml @@ -23,7 +23,7 @@ hardware: benchmark: mode: e2e use_nv_sa_benchmark: false - multi_round: 1 + multi_round: 8 benchmark_ratio: 0.8 streaming: true concurrency_list: '1024' diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT.yaml index 8752c59e1f79..b37216aee27e 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT.yaml @@ -24,7 +24,7 @@ hardware: benchmark: mode: e2e use_nv_sa_benchmark: false - multi_round: 1 + multi_round: 8 benchmark_ratio: 0.8 streaming: true concurrency_list: '1024' From 4d00f3229d196168a8ca913921a3f8d10a3d54ab Mon Sep 17 00:00:00 2001 From: FredricZ-2007 <226039983+fredricz-20070104@users.noreply.github.com> Date: Fri, 6 Mar 2026 06:19:56 +0000 Subject: [PATCH 06/12] split current test to diff concurrency Signed-off-by: FredricZ-2007 <226039983+fredricz-20070104@users.noreply.github.com> --- .../test_lists/qa/llm_perf_multinode.txt | 234 +++++++++++------- ...p16_bs64_eplb0_mtp3_con1024_ccb-NIXL.yaml} | 9 +- ...ep16_bs64_eplb0_mtp3_con1024_ccb-UCX.yaml} | 9 +- ...dep16_bs64_eplb0_mtp3_con512_ccb-NIXL.yaml | 106 ++++++++ ..._dep16_bs64_eplb0_mtp3_con512_ccb-UCX.yaml | 106 ++++++++ ...ep32_bs16_eplb0_mtp3_con512_ccb-NIXL.yaml} | 0 ...dep32_bs16_eplb0_mtp3_con512_ccb-UCX.yaml} | 0 ..._tep8_bs32_eplb0_mtp0_con16_ccb-NIXL.yaml} | 9 +- ...4_tep8_bs32_eplb0_mtp0_con16_ccb-UCX.yaml} | 9 +- ...n4_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL.yaml | 101 ++++++++ ...en4_tep8_bs32_eplb0_mtp0_con1_ccb-UCX.yaml | 101 ++++++++ ...n4_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL.yaml | 101 ++++++++ ...en4_tep8_bs32_eplb0_mtp0_con2_ccb-UCX.yaml | 101 ++++++++ ...4_tep8_bs32_eplb0_mtp0_con32_ccb-NIXL.yaml | 101 ++++++++ ...n4_tep8_bs32_eplb0_mtp0_con32_ccb-UCX.yaml | 101 ++++++++ ...n4_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL.yaml | 101 ++++++++ ...en4_tep8_bs32_eplb0_mtp0_con4_ccb-UCX.yaml | 101 ++++++++ ...n4_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL.yaml | 101 ++++++++ ...en4_tep8_bs32_eplb0_mtp0_con8_ccb-UCX.yaml | 101 ++++++++ ...16_bs128_eplb0_mtp1_con2048_ccb-NIXL.yaml} | 0 ...p16_bs128_eplb0_mtp1_con2048_ccb-UCX.yaml} | 0 ..._tep8_bs32_eplb0_mtp0_con16_ccb-NIXL.yaml} | 10 +- ...1_tep8_bs32_eplb0_mtp0_con16_ccb-UCX.yaml} | 10 +- ...n1_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL.yaml | 91 +++++++ ...en1_tep8_bs32_eplb0_mtp0_con1_ccb-UCX.yaml | 91 +++++++ ...n1_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL.yaml | 91 +++++++ ...en1_tep8_bs32_eplb0_mtp0_con2_ccb-UCX.yaml | 91 +++++++ ...1_tep8_bs32_eplb0_mtp0_con36_ccb-NIXL.yaml | 91 +++++++ ...n1_tep8_bs32_eplb0_mtp0_con36_ccb-UCX.yaml | 91 +++++++ ...n1_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL.yaml | 91 +++++++ ...en1_tep8_bs32_eplb0_mtp0_con4_ccb-UCX.yaml | 91 +++++++ ...n1_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL.yaml | 91 +++++++ ...en1_tep8_bs32_eplb0_mtp0_con8_ccb-UCX.yaml | 91 +++++++ ...n13_tep4_bs1_eplb0_mtp0_con1-Default.yaml} | 0 ...en5_tep4_bs4_eplb0_mtp0_con4-Default.yaml} | 0 ...en6_tep8_bs1_eplb0_mtp3_con1-Default.yaml} | 0 ...en7_tep8_bs1_eplb0_mtp0_con1-Default.yaml} | 0 ...en8_tep4_bs2_eplb0_mtp0_con2-Default.yaml} | 0 ...en8_tep8_bs1_eplb0_mtp0_con1-Default.yaml} | 0 ...n11_tep4_bs2_eplb0_mtp0_con2-Default.yaml} | 0 ...n14_tep4_bs1_eplb0_mtp0_con1-Default.yaml} | 0 ...n1_dep16_bs1_eplb0_mtp3_con1-Default.yaml} | 0 ...en1_dep8_bs4_eplb0_mtp2_con4-Default.yaml} | 0 ...en1_tep8_bs1_eplb0_mtp0_con1-Default.yaml} | 0 ...en1_tep8_bs1_eplb0_mtp3_con1-Default.yaml} | 0 ...en1_tep8_bs2_eplb0_mtp3_con2-Default.yaml} | 0 ...en5_tep8_bs2_eplb0_mtp3_con2-Default.yaml} | 0 ...en7_tep4_bs2_eplb0_mtp2_con2-Default.yaml} | 0 ...en7_tep8_bs1_eplb0_mtp0_con1-Default.yaml} | 0 ...en8_tep4_bs4_eplb0_mtp0_con4-Default.yaml} | 0 ...en7_tep8_bs2_eplb0_mtp3_con2-Default.yaml} | 0 ...n1_dep16_bs8_eplb0_mtp0_con8-Default.yaml} | 0 ...n1_dep32_bs2_eplb0_mtp0_con2-Default.yaml} | 0 ..._dep8_bs16_eplb0_mtp1_con128-Default.yaml} | 0 ..._dep16_bs16_eplb0_mtp0_con16-Default.yaml} | 0 ...n1_dep16_bs8_eplb0_mtp2_con8-Default.yaml} | 0 ...n1_dep32_bs2_eplb0_mtp3_con2-Default.yaml} | 0 ...n1_dep32_bs4_eplb0_mtp0_con4-Default.yaml} | 0 ...dep16_bs16_eplb0_mtp0_con256-Default.yaml} | 0 ..._dep16_bs8_eplb0_mtp3_con128-Default.yaml} | 0 ...1_dep32_bs2_eplb0_mtp3_con64-Default.yaml} | 0 ..._dep32_bs4_eplb0_mtp0_con128-Default.yaml} | 0 ...dep16_bs16_eplb0_mtp1_con256-Default.yaml} | 0 ...dep16_bs32_eplb0_mtp0_con512-Default.yaml} | 0 ...dep16_bs32_eplb0_mtp1_con512-Default.yaml} | 0 ..._dep32_bs4_eplb0_mtp3_con128-Default.yaml} | 0 ..._dep32_bs8_eplb0_mtp0_con256-Default.yaml} | 0 ..._dep32_bs8_eplb0_mtp3_con256-Default.yaml} | 0 ...p32_bs32_eplb0_mtp0_con1024_ccb-NIXL.yaml} | 0 ...ep32_bs32_eplb0_mtp0_con1024_ccb-UCX.yaml} | 0 ..._tep8_bs32_eplb0_mtp0_con16_ccb-NIXL.yaml} | 9 +- ...4_tep8_bs32_eplb0_mtp0_con16_ccb-UCX.yaml} | 9 +- ...n4_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL.yaml | 101 ++++++++ ...en4_tep8_bs32_eplb0_mtp0_con1_ccb-UCX.yaml | 101 ++++++++ ...n4_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL.yaml | 101 ++++++++ ...en4_tep8_bs32_eplb0_mtp0_con2_ccb-UCX.yaml | 101 ++++++++ ...4_tep8_bs32_eplb0_mtp0_con32_ccb-NIXL.yaml | 101 ++++++++ ...n4_tep8_bs32_eplb0_mtp0_con32_ccb-UCX.yaml | 101 ++++++++ ...n4_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL.yaml | 101 ++++++++ ...en4_tep8_bs32_eplb0_mtp0_con4_ccb-UCX.yaml | 101 ++++++++ ...n4_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL.yaml | 101 ++++++++ ...en4_tep8_bs32_eplb0_mtp0_con8_ccb-UCX.yaml | 101 ++++++++ ..._tep8_bs32_eplb0_mtp3_con16_ccb-NIXL.yaml} | 9 +- ...4_tep8_bs32_eplb0_mtp3_con16_ccb-UCX.yaml} | 9 +- ...n4_tep8_bs32_eplb0_mtp3_con1_ccb-NIXL.yaml | 107 ++++++++ ...en4_tep8_bs32_eplb0_mtp3_con1_ccb-UCX.yaml | 107 ++++++++ ...n4_tep8_bs32_eplb0_mtp3_con2_ccb-NIXL.yaml | 107 ++++++++ ...en4_tep8_bs32_eplb0_mtp3_con2_ccb-UCX.yaml | 107 ++++++++ ...4_tep8_bs32_eplb0_mtp3_con32_ccb-NIXL.yaml | 107 ++++++++ ...n4_tep8_bs32_eplb0_mtp3_con32_ccb-UCX.yaml | 107 ++++++++ ...n4_tep8_bs32_eplb0_mtp3_con4_ccb-NIXL.yaml | 107 ++++++++ ...en4_tep8_bs32_eplb0_mtp3_con4_ccb-UCX.yaml | 107 ++++++++ ...n4_tep8_bs32_eplb0_mtp3_con8_ccb-NIXL.yaml | 107 ++++++++ ...en4_tep8_bs32_eplb0_mtp3_con8_ccb-UCX.yaml | 107 ++++++++ ...16_bs128_eplb0_mtp3_con2048_ccb-NIXL.yaml} | 0 ...p16_bs128_eplb0_mtp3_con2048_ccb-UCX.yaml} | 0 ...p32_bs128_eplb0_mtp3_con1024_ccb-UCX.yaml} | 0 ..._tep8_bs16_eplb0_mtp3_con16_ccb-NIXL.yaml} | 9 +- ...3_tep8_bs16_eplb0_mtp3_con16_ccb-UCX.yaml} | 9 +- ...n3_tep8_bs16_eplb0_mtp3_con1_ccb-NIXL.yaml | 107 ++++++++ ...en3_tep8_bs16_eplb0_mtp3_con1_ccb-UCX.yaml | 107 ++++++++ ...n3_tep8_bs16_eplb0_mtp3_con2_ccb-NIXL.yaml | 107 ++++++++ ...en3_tep8_bs16_eplb0_mtp3_con2_ccb-UCX.yaml | 107 ++++++++ ...n3_tep8_bs16_eplb0_mtp3_con4_ccb-NIXL.yaml | 107 ++++++++ ...en3_tep8_bs16_eplb0_mtp3_con4_ccb-UCX.yaml | 107 ++++++++ ...n3_tep8_bs16_eplb0_mtp3_con8_ccb-NIXL.yaml | 107 ++++++++ ...en3_tep8_bs16_eplb0_mtp3_con8_ccb-UCX.yaml | 107 ++++++++ ..._tep8_bs32_eplb0_mtp0_con16_ccb-NIXL.yaml} | 9 +- ...3_tep8_bs32_eplb0_mtp0_con16_ccb-UCX.yaml} | 9 +- ...n3_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL.yaml | 101 ++++++++ ...en3_tep8_bs32_eplb0_mtp0_con1_ccb-UCX.yaml | 101 ++++++++ ...n3_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL.yaml | 101 ++++++++ ...en3_tep8_bs32_eplb0_mtp0_con2_ccb-UCX.yaml | 101 ++++++++ ...3_tep8_bs32_eplb0_mtp0_con32_ccb-NIXL.yaml | 101 ++++++++ ...n3_tep8_bs32_eplb0_mtp0_con32_ccb-UCX.yaml | 101 ++++++++ ...n3_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL.yaml | 101 ++++++++ ...en3_tep8_bs32_eplb0_mtp0_con4_ccb-UCX.yaml | 101 ++++++++ ...n3_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL.yaml | 101 ++++++++ ...en3_tep8_bs32_eplb0_mtp0_con8_ccb-UCX.yaml | 101 ++++++++ ...p16_bs64_eplb0_mtp0_con1024_ccb-NIXL.yaml} | 0 ...ep16_bs64_eplb0_mtp0_con1024_ccb-UCX.yaml} | 0 ...ep32_bs16_eplb0_mtp3_con512_ccb-NIXL.yaml} | 0 ...dep32_bs16_eplb0_mtp3_con512_ccb-UCX.yaml} | 0 ...6_bs64_eplb288_mtp3_con1024_ccb-NIXL.yaml} | 9 +- ...16_bs64_eplb288_mtp3_con1024_ccb-UCX.yaml} | 9 +- ...p16_bs64_eplb288_mtp3_con512_ccb-NIXL.yaml | 114 +++++++++ ...ep16_bs64_eplb288_mtp3_con512_ccb-UCX.yaml | 114 +++++++++ ...32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml} | 0 ...p32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml} | 0 ..._bs128_eplb288_mtp1_con2048_ccb-NIXL.yaml} | 0 ...6_bs128_eplb288_mtp1_con2048_ccb-UCX.yaml} | 0 ...2_bs32_eplb288_mtp0_con1024_ccb-NIXL.yaml} | 0 ...lb288_mtp0_con1024_ccb-NIXL_kv-reuse.yaml} | 0 ...32_bs32_eplb288_mtp0_con1024_ccb-UCX.yaml} | 0 ..._bs128_eplb288_mtp3_con2048_ccb-NIXL.yaml} | 0 ...6_bs128_eplb288_mtp3_con2048_ccb-UCX.yaml} | 0 ...16_eplb288_mtp3_con12288_ccb-DEFAULT.yaml} | 0 ...128_eplb288_mtp3_con1024_ccb-DEFAULT.yaml} | 0 ...6_bs64_eplb288_mtp0_con1024_ccb-NIXL.yaml} | 0 ...16_bs64_eplb288_mtp0_con1024_ccb-UCX.yaml} | 0 ...32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml} | 0 ...p32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml} | 0 ...2_bs32_eplb288_mtp0_con1024_ccb-NIXL.yaml} | 0 ..._bs128_eplb288_mtp3_con2048_ccb-NIXL.yaml} | 0 ...16_eplb288_mtp3_con12288_ccb-DEFAULT.yaml} | 0 ...128_eplb288_mtp3_con1024_ccb-DEFAULT.yaml} | 0 ...6_bs64_eplb288_mtp0_con1024_ccb-NIXL.yaml} | 0 ...32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml} | 0 ...s1024_eplb384_mtp0_con16384_ccb-NIXL.yaml} | 0 ..._bs256_eplb416_mtp0_con8192_ccb-NIXL.yaml} | 0 150 files changed, 6535 insertions(+), 151 deletions(-) rename tests/scripts/perf/disaggregated/{Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_ccb-NIXL.yaml => Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con1024_ccb-NIXL.yaml} (90%) rename tests/scripts/perf/disaggregated/{Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_ccb-UCX.yaml => Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con1024_ccb-UCX.yaml} (90%) create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con512_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con512_ccb-UCX.yaml rename tests/scripts/perf/disaggregated/{Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_ccb-NIXL.yaml => Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_con512_ccb-NIXL.yaml} (100%) rename tests/scripts/perf/disaggregated/{Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_ccb-UCX.yaml => Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_con512_ccb-UCX.yaml} (100%) rename tests/scripts/perf/disaggregated/{Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml => Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con16_ccb-NIXL.yaml} (89%) rename tests/scripts/perf/disaggregated/{Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml => Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con16_ccb-UCX.yaml} (89%) create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con1_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con2_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con32_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con32_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con4_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con8_ccb-UCX.yaml rename tests/scripts/perf/disaggregated/{Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_ccb-NIXL.yaml => Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_con2048_ccb-NIXL.yaml} (100%) rename tests/scripts/perf/disaggregated/{Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_ccb-UCX.yaml => Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_con2048_ccb-UCX.yaml} (100%) rename tests/scripts/perf/disaggregated/{Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml => Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con16_ccb-NIXL.yaml} (88%) rename tests/scripts/perf/disaggregated/{Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml => Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con16_ccb-UCX.yaml} (88%) create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con1_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con2_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con36_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con36_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con4_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con8_ccb-UCX.yaml rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx1_pp4_gen13_tep4_bs1_eplb0_mtp0-Default.yaml => deepseek-r1-fp4_128k8k_ctx1_pp4_gen13_tep4_bs1_eplb0_mtp0_con1-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx1_pp4_gen5_tep4_bs4_eplb0_mtp0-Default.yaml => deepseek-r1-fp4_128k8k_ctx1_pp4_gen5_tep4_bs4_eplb0_mtp0_con4-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx1_pp4_gen6_tep8_bs1_eplb0_mtp3-Default.yaml => deepseek-r1-fp4_128k8k_ctx1_pp4_gen6_tep8_bs1_eplb0_mtp3_con1-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx1_pp4_gen7_tep8_bs1_eplb0_mtp0-Default.yaml => deepseek-r1-fp4_128k8k_ctx1_pp4_gen7_tep8_bs1_eplb0_mtp0_con1-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep4_bs2_eplb0_mtp0-Default.yaml => deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep4_bs2_eplb0_mtp0_con2-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep8_bs1_eplb0_mtp0-Default.yaml => deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep8_bs1_eplb0_mtp0_con1-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx1_pp8_gen11_tep4_bs2_eplb0_mtp0-Default.yaml => deepseek-r1-fp4_128k8k_ctx1_pp8_gen11_tep4_bs2_eplb0_mtp0_con2-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx1_pp8_gen14_tep4_bs1_eplb0_mtp0-Default.yaml => deepseek-r1-fp4_128k8k_ctx1_pp8_gen14_tep4_bs1_eplb0_mtp0_con1-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep16_bs1_eplb0_mtp3-Default.yaml => deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep16_bs1_eplb0_mtp3_con1-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep8_bs4_eplb0_mtp2-Default.yaml => deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep8_bs4_eplb0_mtp2_con4-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp0-Default.yaml => deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp0_con1-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp3-Default.yaml => deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp3_con1-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs2_eplb0_mtp3-Default.yaml => deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs2_eplb0_mtp3_con2-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx1_pp8_gen5_tep8_bs2_eplb0_mtp3-Default.yaml => deepseek-r1-fp4_128k8k_ctx1_pp8_gen5_tep8_bs2_eplb0_mtp3_con2-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep4_bs2_eplb0_mtp2-Default.yaml => deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep4_bs2_eplb0_mtp2_con2-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep8_bs1_eplb0_mtp0-Default.yaml => deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep8_bs1_eplb0_mtp0_con1-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx1_pp8_gen8_tep4_bs4_eplb0_mtp0-Default.yaml => deepseek-r1-fp4_128k8k_ctx1_pp8_gen8_tep4_bs4_eplb0_mtp0_con4-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx2_pp4_gen7_tep8_bs2_eplb0_mtp3-Default.yaml => deepseek-r1-fp4_128k8k_ctx2_pp4_gen7_tep8_bs2_eplb0_mtp3_con2-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep16_bs8_eplb0_mtp0-Default.yaml => deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep16_bs8_eplb0_mtp0_con8-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep32_bs2_eplb0_mtp0-Default.yaml => deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep32_bs2_eplb0_mtp0_con2-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx3_pp4_gen1_dep8_bs16_eplb0_mtp1-Default.yaml => deepseek-r1-fp4_128k8k_ctx3_pp4_gen1_dep8_bs16_eplb0_mtp1_con128-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs16_eplb0_mtp0-Default.yaml => deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs16_eplb0_mtp0_con16-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs8_eplb0_mtp2-Default.yaml => deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs8_eplb0_mtp2_con8-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs2_eplb0_mtp3-Default.yaml => deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs2_eplb0_mtp3_con2-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs4_eplb0_mtp0-Default.yaml => deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs4_eplb0_mtp0_con4-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs16_eplb0_mtp0-Default.yaml => deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs16_eplb0_mtp0_con256-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs8_eplb0_mtp3-Default.yaml => deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs8_eplb0_mtp3_con128-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs2_eplb0_mtp3-Default.yaml => deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs2_eplb0_mtp3_con64-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs4_eplb0_mtp0-Default.yaml => deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs4_eplb0_mtp0_con128-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs16_eplb0_mtp1-Default.yaml => deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs16_eplb0_mtp1_con256-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs32_eplb0_mtp0-Default.yaml => deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs32_eplb0_mtp0_con512-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep16_bs32_eplb0_mtp1-Default.yaml => deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep16_bs32_eplb0_mtp1_con512-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs4_eplb0_mtp3-Default.yaml => deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs4_eplb0_mtp3_con128-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp0-Default.yaml => deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp0_con256-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp3-Default.yaml => deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp3_con256-Default.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb0_mtp0_ccb-NIXL.yaml => deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb0_mtp0_con1024_ccb-NIXL.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb0_mtp0_ccb-UCX.yaml => deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb0_mtp0_con1024_ccb-UCX.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml => deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con16_ccb-NIXL.yaml} (89%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml => deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con16_ccb-UCX.yaml} (89%) create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con1_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con2_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con32_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con32_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con4_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con8_ccb-UCX.yaml rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_ccb-NIXL.yaml => deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con16_ccb-NIXL.yaml} (90%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_ccb-UCX.yaml => deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con16_ccb-UCX.yaml} (90%) create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con1_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con1_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con2_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con2_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con32_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con32_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con4_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con4_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con8_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con8_ccb-UCX.yaml rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp3_ccb-NIXL.yaml => deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp3_con2048_ccb-NIXL.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp3_ccb-UCX.yaml => deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp3_con2048_ccb-UCX.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_8k1k_ctx1_gen1_dep32_bs128_eplb0_mtp3_ccb-UCX.yaml => deepseek-r1-fp4_8k1k_ctx1_gen1_dep32_bs128_eplb0_mtp3_con1024_ccb-UCX.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_ccb-NIXL.yaml => deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con16_ccb-NIXL.yaml} (90%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_ccb-UCX.yaml => deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con16_ccb-UCX.yaml} (90%) create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con1_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con1_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con2_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con2_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con4_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con4_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con8_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con8_ccb-UCX.yaml rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml => deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con16_ccb-NIXL.yaml} (89%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml => deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con16_ccb-UCX.yaml} (89%) create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con1_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con2_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con32_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con32_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con4_ccb-UCX.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con8_ccb-UCX.yaml rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb0_mtp0_ccb-NIXL.yaml => deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb0_mtp0_con1024_ccb-NIXL.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb0_mtp0_ccb-UCX.yaml => deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb0_mtp0_con1024_ccb-UCX.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb0_mtp3_ccb-NIXL.yaml => deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb0_mtp3_con512_ccb-NIXL.yaml} (100%) rename tests/scripts/perf/disaggregated/{deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb0_mtp3_ccb-UCX.yaml => deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb0_mtp3_con512_ccb-UCX.yaml} (100%) rename tests/scripts/perf/disaggregated/{wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-NIXL.yaml => wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-NIXL.yaml} (91%) rename tests/scripts/perf/disaggregated/{wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-UCX.yaml => wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-UCX.yaml} (91%) create mode 100644 tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-NIXL.yaml create mode 100644 tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-UCX.yaml rename tests/scripts/perf/disaggregated/{wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml => wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml} (100%) rename tests/scripts/perf/disaggregated/{wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_ccb-UCX.yaml => wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml} (100%) rename tests/scripts/perf/disaggregated/{wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_ccb-NIXL.yaml => wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-NIXL.yaml} (100%) rename tests/scripts/perf/disaggregated/{wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_ccb-UCX.yaml => wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-UCX.yaml} (100%) rename tests/scripts/perf/disaggregated/{wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL.yaml => wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL.yaml} (100%) rename tests/scripts/perf/disaggregated/{wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL_kv-reuse.yaml => wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL_kv-reuse.yaml} (100%) rename tests/scripts/perf/disaggregated/{wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-UCX.yaml => wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-UCX.yaml} (100%) rename tests/scripts/perf/disaggregated/{wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-NIXL.yaml => wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-NIXL.yaml} (100%) rename tests/scripts/perf/disaggregated/{wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-UCX.yaml => wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-UCX.yaml} (100%) rename tests/scripts/perf/disaggregated/{wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT.yaml => wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_con12288_ccb-DEFAULT.yaml} (100%) rename tests/scripts/perf/disaggregated/{wideep_deepseek-r1-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT.yaml => wideep_deepseek-r1-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_con1024_ccb-DEFAULT.yaml} (100%) rename tests/scripts/perf/disaggregated/{wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL.yaml => wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-NIXL.yaml} (100%) rename tests/scripts/perf/disaggregated/{wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-UCX.yaml => wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-UCX.yaml} (100%) rename tests/scripts/perf/disaggregated/{wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml => wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml} (100%) rename tests/scripts/perf/disaggregated/{wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-UCX.yaml => wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml} (100%) rename tests/scripts/perf/disaggregated/{wideep_deepseek-v32-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL.yaml => wideep_deepseek-v32-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL.yaml} (100%) rename tests/scripts/perf/disaggregated/{wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-NIXL.yaml => wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-NIXL.yaml} (100%) rename tests/scripts/perf/disaggregated/{wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT.yaml => wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_con12288_ccb-DEFAULT.yaml} (100%) rename tests/scripts/perf/disaggregated/{wideep_deepseek-v32-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT.yaml => wideep_deepseek-v32-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_con1024_ccb-DEFAULT.yaml} (100%) rename tests/scripts/perf/disaggregated/{wideep_deepseek-v32-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL.yaml => wideep_deepseek-v32-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-NIXL.yaml} (100%) rename tests/scripts/perf/disaggregated/{wideep_deepseek-v32-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml => wideep_deepseek-v32-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml} (100%) rename tests/scripts/perf/disaggregated/{wideep_kimi-k2-thinking-fp4_1k1k_ctx3_gen1_dep32_bs1024_eplb384_mtp0_ccb-NIXL.yaml => wideep_kimi-k2-thinking-fp4_1k1k_ctx3_gen1_dep32_bs1024_eplb384_mtp0_con16384_ccb-NIXL.yaml} (100%) rename tests/scripts/perf/disaggregated/{wideep_kimi-k2-thinking-fp4_8k1k_ctx8_gen1_dep32_bs256_eplb416_mtp0_ccb-NIXL.yaml => wideep_kimi-k2-thinking-fp4_8k1k_ctx8_gen1_dep32_bs256_eplb416_mtp0_con8192_ccb-NIXL.yaml} (100%) diff --git a/tests/integration/test_lists/qa/llm_perf_multinode.txt b/tests/integration/test_lists/qa/llm_perf_multinode.txt index 5d048aa20495..5499073e29d3 100644 --- a/tests/integration/test_lists/qa/llm_perf_multinode.txt +++ b/tests/integration/test_lists/qa/llm_perf_multinode.txt @@ -1,98 +1,162 @@ # disagg multi-node # GB200 + GB300 supported cases -# perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_ccb-NIXL] -# perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_ccb-UCX] -# perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_ccb-NIXL] -# perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_ccb-UCX] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-UCX] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_ccb-UCX] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb0_mtp0_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb0_mtp0_ccb-UCX] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-UCX] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_ccb-UCX] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp3_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp3_ccb-UCX] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen1_dep32_bs128_eplb0_mtp3_ccb-UCX] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_ccb-UCX] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_ccb-UCX] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb0_mtp0_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb0_mtp0_ccb-UCX] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb0_mtp3_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb0_mtp3_ccb-UCX] +# perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con512_ccb-NIXL] +# perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con1024_ccb-NIXL] +# perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con512_ccb-UCX] +# perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con1024_ccb-UCX] +# perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_con512_ccb-NIXL] +# perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_con512_ccb-UCX] +# perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_con2048_ccb-NIXL] +# perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_con2048_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con16_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con32_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con1_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con2_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con4_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con8_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con16_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con32_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con16_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con36_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con1_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con2_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con4_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con8_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con16_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con36_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb0_mtp0_con1024_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb0_mtp0_con1024_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con16_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con32_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con1_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con2_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con4_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con8_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con16_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con32_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con1_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con2_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con4_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con8_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con16_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con32_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con1_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con2_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con4_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con8_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con16_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con32_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp3_con2048_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp3_con2048_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen1_dep32_bs128_eplb0_mtp3_con1024_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con1_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con2_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con4_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con8_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con16_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con1_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con2_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con4_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con8_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con16_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con16_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con32_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con1_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con2_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con4_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con8_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con16_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con32_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb0_mtp0_con1024_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb0_mtp0_con1024_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb0_mtp3_con512_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb0_mtp3_con512_ccb-UCX] # GB200 supported cases -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen11_tep4_bs2_eplb0_mtp0-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen14_tep4_bs1_eplb0_mtp0-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep16_bs1_eplb0_mtp3-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep8_bs4_eplb0_mtp2-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp0-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp3-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs2_eplb0_mtp3-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen5_tep8_bs2_eplb0_mtp3-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep4_bs2_eplb0_mtp2-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep8_bs1_eplb0_mtp0-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen8_tep4_bs4_eplb0_mtp0-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep16_bs8_eplb0_mtp0-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep32_bs2_eplb0_mtp0-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs16_eplb0_mtp0-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs8_eplb0_mtp2-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs2_eplb0_mtp3-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs4_eplb0_mtp0-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen11_tep4_bs2_eplb0_mtp0_con2-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen14_tep4_bs1_eplb0_mtp0_con1-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep16_bs1_eplb0_mtp3_con1-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep8_bs4_eplb0_mtp2_con4-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp0_con1-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp3_con1-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs2_eplb0_mtp3_con2-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen5_tep8_bs2_eplb0_mtp3_con2-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep4_bs2_eplb0_mtp2_con2-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep8_bs1_eplb0_mtp0_con1-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp8_gen8_tep4_bs4_eplb0_mtp0_con4-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep16_bs8_eplb0_mtp0_con8-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep32_bs2_eplb0_mtp0_con2-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs16_eplb0_mtp0_con16-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs8_eplb0_mtp2_con8-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs2_eplb0_mtp3_con2-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs4_eplb0_mtp0_con4-Default] # GB300 supported cases -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp4_gen13_tep4_bs1_eplb0_mtp0-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp4_gen5_tep4_bs4_eplb0_mtp0-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp4_gen6_tep8_bs1_eplb0_mtp3-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp4_gen7_tep8_bs1_eplb0_mtp0-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep4_bs2_eplb0_mtp0-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep8_bs1_eplb0_mtp0-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx2_pp4_gen7_tep8_bs2_eplb0_mtp3-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx3_pp4_gen1_dep8_bs16_eplb0_mtp1-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs16_eplb0_mtp0-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs8_eplb0_mtp3-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs2_eplb0_mtp3-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs4_eplb0_mtp0-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs16_eplb0_mtp1-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs32_eplb0_mtp0-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep16_bs32_eplb0_mtp1-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs4_eplb0_mtp3-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp0-Default] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp3-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp4_gen13_tep4_bs1_eplb0_mtp0_con1-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp4_gen5_tep4_bs4_eplb0_mtp0_con4-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp4_gen6_tep8_bs1_eplb0_mtp3_con1-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp4_gen7_tep8_bs1_eplb0_mtp0_con1-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep4_bs2_eplb0_mtp0_con2-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep8_bs1_eplb0_mtp0_con1-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx2_pp4_gen7_tep8_bs2_eplb0_mtp3_con2-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx3_pp4_gen1_dep8_bs16_eplb0_mtp1_con128-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs16_eplb0_mtp0_con256-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs8_eplb0_mtp3_con128-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs2_eplb0_mtp3_con64-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs4_eplb0_mtp0_con128-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs16_eplb0_mtp1_con256-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs32_eplb0_mtp0_con512-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep16_bs32_eplb0_mtp1_con512-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs4_eplb0_mtp3_con128-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp0_con256-Default] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp3_con256-Default] # wideep multi-node # GB200 + GB300 supported cases -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-UCX] -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_ccb-UCX] -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_ccb-UCX] -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL_kv-reuse] -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-UCX] -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-UCX] -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-wideep_deepseek-r1-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT] -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-UCX] -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-UCX] -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-v32-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT] -perf/test_perf_sanity.py::test_e2e[disagg-e2e-wideep_deepseek-v32-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT] -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-v32-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-v32-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_kimi-k2-thinking-fp4_1k1k_ctx3_gen1_dep32_bs1024_eplb384_mtp0_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_kimi-k2-thinking-fp4_8k1k_ctx8_gen1_dep32_bs256_eplb416_mtp0_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL_kv-reuse] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_con12288_ccb-DEFAULT] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-wideep_deepseek-r1-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_con1024_ccb-DEFAULT] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-v32-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_con12288_ccb-DEFAULT] +perf/test_perf_sanity.py::test_e2e[disagg-e2e-wideep_deepseek-v32-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_con1024_ccb-DEFAULT] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-v32-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-v32-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_kimi-k2-thinking-fp4_1k1k_ctx3_gen1_dep32_bs1024_eplb384_mtp0_con16384_ccb-NIXL] +perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_kimi-k2-thinking-fp4_8k1k_ctx8_gen1_dep32_bs256_eplb416_mtp0_con8192_ccb-NIXL] # GB200 supported cases # GB300 supported cases diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con1024_ccb-NIXL.yaml similarity index 90% rename from tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_ccb-NIXL.yaml rename to tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con1024_ccb-NIXL.yaml index 2f94667a536c..1398cf8cc87e 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con1024_ccb-NIXL.yaml @@ -13,7 +13,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: e2e @@ -21,7 +21,7 @@ benchmark: multi_round: 8 benchmark_ratio: 0.8 streaming: true - concurrency_list: 512 1024 + concurrency_list: '1024' input_length: 1024 output_length: 1024 dataset_file: @@ -36,8 +36,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con1024_ccb-UCX.yaml similarity index 90% rename from tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_ccb-UCX.yaml rename to tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con1024_ccb-UCX.yaml index fba25771bb18..ac5fceb71884 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con1024_ccb-UCX.yaml @@ -13,7 +13,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: e2e @@ -21,7 +21,7 @@ benchmark: multi_round: 8 benchmark_ratio: 0.8 streaming: true - concurrency_list: 512 1024 + concurrency_list: '1024' input_length: 1024 output_length: 1024 dataset_file: @@ -36,8 +36,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con512_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con512_ccb-NIXL.yaml new file mode 100644 index 000000000000..fa99044fce7b --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con512_ccb-NIXL.yaml @@ -0,0 +1,106 @@ +metadata: + model_name: Qwen3-235B-A22B-FP4 + precision: fp4 + model_dir_name: Qwen3-235B-A22B-FP4 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '512' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 64 + max_num_tokens: 256 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: WIDEEP + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con512_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con512_ccb-UCX.yaml new file mode 100644 index 000000000000..e8475a5df2f4 --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con512_ccb-UCX.yaml @@ -0,0 +1,106 @@ +metadata: + model_name: Qwen3-235B-A22B-FP4 + precision: fp4 + model_dir_name: Qwen3-235B-A22B-FP4 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '512' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + pipeline_parallel_size: 1 + max_batch_size: 64 + max_num_tokens: 256 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: WIDEEP + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_con512_ccb-NIXL.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_ccb-NIXL.yaml rename to tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_con512_ccb-NIXL.yaml diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_con512_ccb-UCX.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_ccb-UCX.yaml rename to tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_con512_ccb-UCX.yaml diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con16_ccb-NIXL.yaml similarity index 89% rename from tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml rename to tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con16_ccb-NIXL.yaml index 1ff8177d2348..72ff6a3f579a 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con16_ccb-NIXL.yaml @@ -13,7 +13,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: e2e @@ -21,7 +21,7 @@ benchmark: multi_round: 8 benchmark_ratio: 0.8 streaming: true - concurrency_list: 1 2 4 8 16 32 + concurrency_list: '16' input_length: 1024 output_length: 1024 dataset_file: @@ -36,8 +36,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con16_ccb-UCX.yaml similarity index 89% rename from tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml rename to tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con16_ccb-UCX.yaml index 0437408d4317..c158ae77ba08 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con16_ccb-UCX.yaml @@ -13,7 +13,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: e2e @@ -21,7 +21,7 @@ benchmark: multi_round: 8 benchmark_ratio: 0.8 streaming: true - concurrency_list: 1 2 4 8 16 32 + concurrency_list: '16' input_length: 1024 output_length: 1024 dataset_file: @@ -36,8 +36,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL.yaml new file mode 100644 index 000000000000..7c77cdf5e035 --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL.yaml @@ -0,0 +1,101 @@ +metadata: + model_name: Qwen3-235B-A22B-FP4 + precision: fp4 + model_dir_name: Qwen3-235B-A22B-FP4 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con1_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con1_ccb-UCX.yaml new file mode 100644 index 000000000000..d978c98e1922 --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con1_ccb-UCX.yaml @@ -0,0 +1,101 @@ +metadata: + model_name: Qwen3-235B-A22B-FP4 + precision: fp4 + model_dir_name: Qwen3-235B-A22B-FP4 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL.yaml new file mode 100644 index 000000000000..224d3b9b659d --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL.yaml @@ -0,0 +1,101 @@ +metadata: + model_name: Qwen3-235B-A22B-FP4 + precision: fp4 + model_dir_name: Qwen3-235B-A22B-FP4 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '2' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con2_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con2_ccb-UCX.yaml new file mode 100644 index 000000000000..347b8645b1c0 --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con2_ccb-UCX.yaml @@ -0,0 +1,101 @@ +metadata: + model_name: Qwen3-235B-A22B-FP4 + precision: fp4 + model_dir_name: Qwen3-235B-A22B-FP4 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '2' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con32_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con32_ccb-NIXL.yaml new file mode 100644 index 000000000000..1a92f6c81925 --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con32_ccb-NIXL.yaml @@ -0,0 +1,101 @@ +metadata: + model_name: Qwen3-235B-A22B-FP4 + precision: fp4 + model_dir_name: Qwen3-235B-A22B-FP4 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '32' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con32_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con32_ccb-UCX.yaml new file mode 100644 index 000000000000..8d09c7ec0d14 --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con32_ccb-UCX.yaml @@ -0,0 +1,101 @@ +metadata: + model_name: Qwen3-235B-A22B-FP4 + precision: fp4 + model_dir_name: Qwen3-235B-A22B-FP4 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '32' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL.yaml new file mode 100644 index 000000000000..5f3cbd8ea60b --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL.yaml @@ -0,0 +1,101 @@ +metadata: + model_name: Qwen3-235B-A22B-FP4 + precision: fp4 + model_dir_name: Qwen3-235B-A22B-FP4 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '4' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con4_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con4_ccb-UCX.yaml new file mode 100644 index 000000000000..270e6a3b96e8 --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con4_ccb-UCX.yaml @@ -0,0 +1,101 @@ +metadata: + model_name: Qwen3-235B-A22B-FP4 + precision: fp4 + model_dir_name: Qwen3-235B-A22B-FP4 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '4' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL.yaml new file mode 100644 index 000000000000..18e0c5bd4da3 --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL.yaml @@ -0,0 +1,101 @@ +metadata: + model_name: Qwen3-235B-A22B-FP4 + precision: fp4 + model_dir_name: Qwen3-235B-A22B-FP4 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '8' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con8_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con8_ccb-UCX.yaml new file mode 100644 index 000000000000..e7c3063804d5 --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con8_ccb-UCX.yaml @@ -0,0 +1,101 @@ +metadata: + model_name: Qwen3-235B-A22B-FP4 + precision: fp4 + model_dir_name: Qwen3-235B-A22B-FP4 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '8' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_con2048_ccb-NIXL.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_ccb-NIXL.yaml rename to tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_con2048_ccb-NIXL.yaml diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_con2048_ccb-UCX.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_ccb-UCX.yaml rename to tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_con2048_ccb-UCX.yaml diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con16_ccb-NIXL.yaml similarity index 88% rename from tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml rename to tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con16_ccb-NIXL.yaml index 7f4c66a24f90..49af3a3259d3 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con16_ccb-NIXL.yaml @@ -1,4 +1,3 @@ -# nvbugs: 5561153 metadata: model_name: Qwen3-235B-A22B-FP8 precision: fp8 @@ -14,7 +13,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: e2e @@ -22,7 +21,7 @@ benchmark: multi_round: 8 benchmark_ratio: 0.8 streaming: true - concurrency_list: 1 2 4 8 16 36 + concurrency_list: '16' input_length: 1024 output_length: 1024 dataset_file: @@ -37,8 +36,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con16_ccb-UCX.yaml similarity index 88% rename from tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml rename to tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con16_ccb-UCX.yaml index b6e6e3d82641..9ee8026bbefb 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con16_ccb-UCX.yaml @@ -1,4 +1,3 @@ -# nvbugs: 5561153 metadata: model_name: Qwen3-235B-A22B-FP8 precision: fp8 @@ -14,7 +13,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: e2e @@ -22,7 +21,7 @@ benchmark: multi_round: 8 benchmark_ratio: 0.8 streaming: true - concurrency_list: 1 2 4 8 16 36 + concurrency_list: '16' input_length: 1024 output_length: 1024 dataset_file: @@ -37,8 +36,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL.yaml new file mode 100644 index 000000000000..321cbf911a0b --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL.yaml @@ -0,0 +1,91 @@ +metadata: + model_name: Qwen3-235B-A22B-FP8 + precision: fp8 + model_dir_name: Qwen3-235B-A22B-FP8 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 64 + max_num_tokens: 2048 + max_seq_len: 2051 + cuda_graph_config: + enable_padding: true + max_batch_size: 128 + print_iter_log: true + kv_cache_config: + enable_block_reuse: true + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 2048 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + disable_overlap_scheduler: false + ctx: + max_batch_size: 32 + max_num_tokens: 2048 + max_seq_len: 2051 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: true + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 2048 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con1_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con1_ccb-UCX.yaml new file mode 100644 index 000000000000..7a7fb603d08c --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con1_ccb-UCX.yaml @@ -0,0 +1,91 @@ +metadata: + model_name: Qwen3-235B-A22B-FP8 + precision: fp8 + model_dir_name: Qwen3-235B-A22B-FP8 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 64 + max_num_tokens: 2048 + max_seq_len: 2051 + cuda_graph_config: + enable_padding: true + max_batch_size: 128 + print_iter_log: true + kv_cache_config: + enable_block_reuse: true + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 2048 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + disable_overlap_scheduler: false + ctx: + max_batch_size: 32 + max_num_tokens: 2048 + max_seq_len: 2051 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: true + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 2048 + backend: UCX diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL.yaml new file mode 100644 index 000000000000..ccda3d3e5886 --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL.yaml @@ -0,0 +1,91 @@ +metadata: + model_name: Qwen3-235B-A22B-FP8 + precision: fp8 + model_dir_name: Qwen3-235B-A22B-FP8 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '2' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 64 + max_num_tokens: 2048 + max_seq_len: 2051 + cuda_graph_config: + enable_padding: true + max_batch_size: 128 + print_iter_log: true + kv_cache_config: + enable_block_reuse: true + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 2048 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + disable_overlap_scheduler: false + ctx: + max_batch_size: 32 + max_num_tokens: 2048 + max_seq_len: 2051 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: true + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 2048 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con2_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con2_ccb-UCX.yaml new file mode 100644 index 000000000000..86ad52efb5a2 --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con2_ccb-UCX.yaml @@ -0,0 +1,91 @@ +metadata: + model_name: Qwen3-235B-A22B-FP8 + precision: fp8 + model_dir_name: Qwen3-235B-A22B-FP8 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '2' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 64 + max_num_tokens: 2048 + max_seq_len: 2051 + cuda_graph_config: + enable_padding: true + max_batch_size: 128 + print_iter_log: true + kv_cache_config: + enable_block_reuse: true + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 2048 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + disable_overlap_scheduler: false + ctx: + max_batch_size: 32 + max_num_tokens: 2048 + max_seq_len: 2051 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: true + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 2048 + backend: UCX diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con36_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con36_ccb-NIXL.yaml new file mode 100644 index 000000000000..d4f58c3cb47c --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con36_ccb-NIXL.yaml @@ -0,0 +1,91 @@ +metadata: + model_name: Qwen3-235B-A22B-FP8 + precision: fp8 + model_dir_name: Qwen3-235B-A22B-FP8 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '36' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 64 + max_num_tokens: 2048 + max_seq_len: 2051 + cuda_graph_config: + enable_padding: true + max_batch_size: 128 + print_iter_log: true + kv_cache_config: + enable_block_reuse: true + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 2048 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + disable_overlap_scheduler: false + ctx: + max_batch_size: 32 + max_num_tokens: 2048 + max_seq_len: 2051 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: true + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 2048 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con36_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con36_ccb-UCX.yaml new file mode 100644 index 000000000000..f10b8f473470 --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con36_ccb-UCX.yaml @@ -0,0 +1,91 @@ +metadata: + model_name: Qwen3-235B-A22B-FP8 + precision: fp8 + model_dir_name: Qwen3-235B-A22B-FP8 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '36' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 64 + max_num_tokens: 2048 + max_seq_len: 2051 + cuda_graph_config: + enable_padding: true + max_batch_size: 128 + print_iter_log: true + kv_cache_config: + enable_block_reuse: true + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 2048 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + disable_overlap_scheduler: false + ctx: + max_batch_size: 32 + max_num_tokens: 2048 + max_seq_len: 2051 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: true + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 2048 + backend: UCX diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL.yaml new file mode 100644 index 000000000000..33a3c60b63c7 --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL.yaml @@ -0,0 +1,91 @@ +metadata: + model_name: Qwen3-235B-A22B-FP8 + precision: fp8 + model_dir_name: Qwen3-235B-A22B-FP8 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '4' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 64 + max_num_tokens: 2048 + max_seq_len: 2051 + cuda_graph_config: + enable_padding: true + max_batch_size: 128 + print_iter_log: true + kv_cache_config: + enable_block_reuse: true + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 2048 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + disable_overlap_scheduler: false + ctx: + max_batch_size: 32 + max_num_tokens: 2048 + max_seq_len: 2051 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: true + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 2048 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con4_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con4_ccb-UCX.yaml new file mode 100644 index 000000000000..e43c2ad1d011 --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con4_ccb-UCX.yaml @@ -0,0 +1,91 @@ +metadata: + model_name: Qwen3-235B-A22B-FP8 + precision: fp8 + model_dir_name: Qwen3-235B-A22B-FP8 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '4' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 64 + max_num_tokens: 2048 + max_seq_len: 2051 + cuda_graph_config: + enable_padding: true + max_batch_size: 128 + print_iter_log: true + kv_cache_config: + enable_block_reuse: true + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 2048 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + disable_overlap_scheduler: false + ctx: + max_batch_size: 32 + max_num_tokens: 2048 + max_seq_len: 2051 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: true + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 2048 + backend: UCX diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL.yaml new file mode 100644 index 000000000000..db8ca23defa9 --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL.yaml @@ -0,0 +1,91 @@ +metadata: + model_name: Qwen3-235B-A22B-FP8 + precision: fp8 + model_dir_name: Qwen3-235B-A22B-FP8 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '8' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 64 + max_num_tokens: 2048 + max_seq_len: 2051 + cuda_graph_config: + enable_padding: true + max_batch_size: 128 + print_iter_log: true + kv_cache_config: + enable_block_reuse: true + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 2048 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + disable_overlap_scheduler: false + ctx: + max_batch_size: 32 + max_num_tokens: 2048 + max_seq_len: 2051 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: true + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 2048 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con8_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con8_ccb-UCX.yaml new file mode 100644 index 000000000000..4d4958076cc0 --- /dev/null +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con8_ccb-UCX.yaml @@ -0,0 +1,91 @@ +metadata: + model_name: Qwen3-235B-A22B-FP8 + precision: fp8 + model_dir_name: Qwen3-235B-A22B-FP8 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '8' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 64 + max_num_tokens: 2048 + max_seq_len: 2051 + cuda_graph_config: + enable_padding: true + max_batch_size: 128 + print_iter_log: true + kv_cache_config: + enable_block_reuse: true + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 2048 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + disable_overlap_scheduler: false + ctx: + max_batch_size: 32 + max_num_tokens: 2048 + max_seq_len: 2051 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: true + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 2048 + backend: UCX diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen13_tep4_bs1_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen13_tep4_bs1_eplb0_mtp0_con1-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen13_tep4_bs1_eplb0_mtp0-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen13_tep4_bs1_eplb0_mtp0_con1-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen5_tep4_bs4_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen5_tep4_bs4_eplb0_mtp0_con4-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen5_tep4_bs4_eplb0_mtp0-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen5_tep4_bs4_eplb0_mtp0_con4-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen6_tep8_bs1_eplb0_mtp3-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen6_tep8_bs1_eplb0_mtp3_con1-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen6_tep8_bs1_eplb0_mtp3-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen6_tep8_bs1_eplb0_mtp3_con1-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen7_tep8_bs1_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen7_tep8_bs1_eplb0_mtp0_con1-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen7_tep8_bs1_eplb0_mtp0-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen7_tep8_bs1_eplb0_mtp0_con1-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep4_bs2_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep4_bs2_eplb0_mtp0_con2-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep4_bs2_eplb0_mtp0-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep4_bs2_eplb0_mtp0_con2-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep8_bs1_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep8_bs1_eplb0_mtp0_con1-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep8_bs1_eplb0_mtp0-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep8_bs1_eplb0_mtp0_con1-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen11_tep4_bs2_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen11_tep4_bs2_eplb0_mtp0_con2-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen11_tep4_bs2_eplb0_mtp0-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen11_tep4_bs2_eplb0_mtp0_con2-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen14_tep4_bs1_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen14_tep4_bs1_eplb0_mtp0_con1-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen14_tep4_bs1_eplb0_mtp0-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen14_tep4_bs1_eplb0_mtp0_con1-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep16_bs1_eplb0_mtp3-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep16_bs1_eplb0_mtp3_con1-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep16_bs1_eplb0_mtp3-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep16_bs1_eplb0_mtp3_con1-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep8_bs4_eplb0_mtp2-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep8_bs4_eplb0_mtp2_con4-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep8_bs4_eplb0_mtp2-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep8_bs4_eplb0_mtp2_con4-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp0_con1-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp0-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp0_con1-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp3-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp3_con1-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp3-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp3_con1-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs2_eplb0_mtp3-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs2_eplb0_mtp3_con2-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs2_eplb0_mtp3-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs2_eplb0_mtp3_con2-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen5_tep8_bs2_eplb0_mtp3-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen5_tep8_bs2_eplb0_mtp3_con2-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen5_tep8_bs2_eplb0_mtp3-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen5_tep8_bs2_eplb0_mtp3_con2-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep4_bs2_eplb0_mtp2-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep4_bs2_eplb0_mtp2_con2-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep4_bs2_eplb0_mtp2-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep4_bs2_eplb0_mtp2_con2-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep8_bs1_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep8_bs1_eplb0_mtp0_con1-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep8_bs1_eplb0_mtp0-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep8_bs1_eplb0_mtp0_con1-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen8_tep4_bs4_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen8_tep4_bs4_eplb0_mtp0_con4-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen8_tep4_bs4_eplb0_mtp0-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen8_tep4_bs4_eplb0_mtp0_con4-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp4_gen7_tep8_bs2_eplb0_mtp3-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp4_gen7_tep8_bs2_eplb0_mtp3_con2-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp4_gen7_tep8_bs2_eplb0_mtp3-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp4_gen7_tep8_bs2_eplb0_mtp3_con2-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep16_bs8_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep16_bs8_eplb0_mtp0_con8-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep16_bs8_eplb0_mtp0-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep16_bs8_eplb0_mtp0_con8-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep32_bs2_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep32_bs2_eplb0_mtp0_con2-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep32_bs2_eplb0_mtp0-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep32_bs2_eplb0_mtp0_con2-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp4_gen1_dep8_bs16_eplb0_mtp1-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp4_gen1_dep8_bs16_eplb0_mtp1_con128-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp4_gen1_dep8_bs16_eplb0_mtp1-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp4_gen1_dep8_bs16_eplb0_mtp1_con128-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs16_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs16_eplb0_mtp0_con16-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs16_eplb0_mtp0-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs16_eplb0_mtp0_con16-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs8_eplb0_mtp2-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs8_eplb0_mtp2_con8-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs8_eplb0_mtp2-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs8_eplb0_mtp2_con8-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs2_eplb0_mtp3-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs2_eplb0_mtp3_con2-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs2_eplb0_mtp3-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs2_eplb0_mtp3_con2-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs4_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs4_eplb0_mtp0_con4-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs4_eplb0_mtp0-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs4_eplb0_mtp0_con4-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs16_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs16_eplb0_mtp0_con256-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs16_eplb0_mtp0-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs16_eplb0_mtp0_con256-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs8_eplb0_mtp3-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs8_eplb0_mtp3_con128-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs8_eplb0_mtp3-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs8_eplb0_mtp3_con128-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs2_eplb0_mtp3-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs2_eplb0_mtp3_con64-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs2_eplb0_mtp3-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs2_eplb0_mtp3_con64-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs4_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs4_eplb0_mtp0_con128-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs4_eplb0_mtp0-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs4_eplb0_mtp0_con128-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs16_eplb0_mtp1-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs16_eplb0_mtp1_con256-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs16_eplb0_mtp1-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs16_eplb0_mtp1_con256-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs32_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs32_eplb0_mtp0_con512-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs32_eplb0_mtp0-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs32_eplb0_mtp0_con512-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep16_bs32_eplb0_mtp1-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep16_bs32_eplb0_mtp1_con512-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep16_bs32_eplb0_mtp1-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep16_bs32_eplb0_mtp1_con512-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs4_eplb0_mtp3-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs4_eplb0_mtp3_con128-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs4_eplb0_mtp3-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs4_eplb0_mtp3_con128-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp0-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp0_con256-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp0-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp0_con256-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp3-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp3_con256-Default.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp3-Default.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp3_con256-Default.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb0_mtp0_con1024_ccb-NIXL.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb0_mtp0_ccb-NIXL.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb0_mtp0_con1024_ccb-NIXL.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb0_mtp0_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb0_mtp0_con1024_ccb-UCX.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb0_mtp0_ccb-UCX.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb0_mtp0_con1024_ccb-UCX.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con16_ccb-NIXL.yaml similarity index 89% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con16_ccb-NIXL.yaml index 9f8ec3a9f956..e837bd894f6d 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con16_ccb-NIXL.yaml @@ -13,7 +13,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: e2e @@ -21,7 +21,7 @@ benchmark: multi_round: 8 benchmark_ratio: 0.8 streaming: true - concurrency_list: 1 2 4 8 16 32 + concurrency_list: '16' input_length: 1024 output_length: 1024 dataset_file: @@ -36,8 +36,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con16_ccb-UCX.yaml similarity index 89% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con16_ccb-UCX.yaml index 6260fc5b8c0e..7ff05e15ed5a 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con16_ccb-UCX.yaml @@ -13,7 +13,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: e2e @@ -21,7 +21,7 @@ benchmark: multi_round: 8 benchmark_ratio: 0.8 streaming: true - concurrency_list: 1 2 4 8 16 32 + concurrency_list: '16' input_length: 1024 output_length: 1024 dataset_file: @@ -36,8 +36,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL.yaml new file mode 100644 index 000000000000..dcbdf85cb14c --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL.yaml @@ -0,0 +1,101 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con1_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con1_ccb-UCX.yaml new file mode 100644 index 000000000000..fde80dab2da1 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con1_ccb-UCX.yaml @@ -0,0 +1,101 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL.yaml new file mode 100644 index 000000000000..8fc9a9d12553 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL.yaml @@ -0,0 +1,101 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '2' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con2_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con2_ccb-UCX.yaml new file mode 100644 index 000000000000..50d674fb79b2 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con2_ccb-UCX.yaml @@ -0,0 +1,101 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '2' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con32_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con32_ccb-NIXL.yaml new file mode 100644 index 000000000000..2adc8e2303b5 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con32_ccb-NIXL.yaml @@ -0,0 +1,101 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '32' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con32_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con32_ccb-UCX.yaml new file mode 100644 index 000000000000..0126bc21acea --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con32_ccb-UCX.yaml @@ -0,0 +1,101 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '32' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL.yaml new file mode 100644 index 000000000000..c718acbe9e32 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL.yaml @@ -0,0 +1,101 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '4' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con4_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con4_ccb-UCX.yaml new file mode 100644 index 000000000000..b5fa6ccf3fef --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con4_ccb-UCX.yaml @@ -0,0 +1,101 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '4' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL.yaml new file mode 100644 index 000000000000..0f04f4cbea67 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL.yaml @@ -0,0 +1,101 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '8' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con8_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con8_ccb-UCX.yaml new file mode 100644 index 000000000000..55b2c50f50d3 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con8_ccb-UCX.yaml @@ -0,0 +1,101 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '8' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con16_ccb-NIXL.yaml similarity index 90% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_ccb-NIXL.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con16_ccb-NIXL.yaml index e3b9d07ff42b..f5cfc62133d5 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con16_ccb-NIXL.yaml @@ -13,7 +13,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: e2e @@ -21,7 +21,7 @@ benchmark: multi_round: 8 benchmark_ratio: 0.8 streaming: true - concurrency_list: 1 2 4 8 16 32 + concurrency_list: '16' input_length: 1024 output_length: 1024 dataset_file: @@ -36,8 +36,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con16_ccb-UCX.yaml similarity index 90% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_ccb-UCX.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con16_ccb-UCX.yaml index b483ce1c8c6a..ab3dcc0f94d9 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con16_ccb-UCX.yaml @@ -13,7 +13,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: e2e @@ -21,7 +21,7 @@ benchmark: multi_round: 8 benchmark_ratio: 0.8 streaming: true - concurrency_list: 1 2 4 8 16 32 + concurrency_list: '16' input_length: 1024 output_length: 1024 dataset_file: @@ -36,8 +36,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con1_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con1_ccb-NIXL.yaml new file mode 100644 index 000000000000..c315d8997679 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con1_ccb-NIXL.yaml @@ -0,0 +1,107 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con1_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con1_ccb-UCX.yaml new file mode 100644 index 000000000000..6bb1b1e3738a --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con1_ccb-UCX.yaml @@ -0,0 +1,107 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con2_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con2_ccb-NIXL.yaml new file mode 100644 index 000000000000..09da6f76da3d --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con2_ccb-NIXL.yaml @@ -0,0 +1,107 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '2' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con2_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con2_ccb-UCX.yaml new file mode 100644 index 000000000000..e87ca8c78676 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con2_ccb-UCX.yaml @@ -0,0 +1,107 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '2' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con32_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con32_ccb-NIXL.yaml new file mode 100644 index 000000000000..33ab0317ec61 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con32_ccb-NIXL.yaml @@ -0,0 +1,107 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '32' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con32_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con32_ccb-UCX.yaml new file mode 100644 index 000000000000..2e57bf2e459f --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con32_ccb-UCX.yaml @@ -0,0 +1,107 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '32' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con4_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con4_ccb-NIXL.yaml new file mode 100644 index 000000000000..49b53a8aa7eb --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con4_ccb-NIXL.yaml @@ -0,0 +1,107 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '4' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con4_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con4_ccb-UCX.yaml new file mode 100644 index 000000000000..64100cb5be91 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con4_ccb-UCX.yaml @@ -0,0 +1,107 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '4' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con8_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con8_ccb-NIXL.yaml new file mode 100644 index 000000000000..ec664924f8fd --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con8_ccb-NIXL.yaml @@ -0,0 +1,107 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '8' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con8_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con8_ccb-UCX.yaml new file mode 100644 index 000000000000..bec518b6cdf4 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp3_con8_ccb-UCX.yaml @@ -0,0 +1,107 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '8' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 4 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp3_con2048_ccb-NIXL.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp3_ccb-NIXL.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp3_con2048_ccb-NIXL.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp3_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp3_con2048_ccb-UCX.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp3_ccb-UCX.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp3_con2048_ccb-UCX.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen1_dep32_bs128_eplb0_mtp3_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen1_dep32_bs128_eplb0_mtp3_con1024_ccb-UCX.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen1_dep32_bs128_eplb0_mtp3_ccb-UCX.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen1_dep32_bs128_eplb0_mtp3_con1024_ccb-UCX.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con16_ccb-NIXL.yaml similarity index 90% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_ccb-NIXL.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con16_ccb-NIXL.yaml index 6ff591400924..8ac786dd9754 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con16_ccb-NIXL.yaml @@ -13,7 +13,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: e2e @@ -21,7 +21,7 @@ benchmark: multi_round: 8 benchmark_ratio: 0.8 streaming: true - concurrency_list: 1 2 4 8 16 + concurrency_list: '16' input_length: 8192 output_length: 1024 dataset_file: @@ -36,8 +36,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con16_ccb-UCX.yaml similarity index 90% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_ccb-UCX.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con16_ccb-UCX.yaml index 7667c7903adc..d89f0fb2d409 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con16_ccb-UCX.yaml @@ -13,7 +13,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: e2e @@ -21,7 +21,7 @@ benchmark: multi_round: 8 benchmark_ratio: 0.8 streaming: true - concurrency_list: 1 2 4 8 16 + concurrency_list: '16' input_length: 8192 output_length: 1024 dataset_file: @@ -36,8 +36,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con1_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con1_ccb-NIXL.yaml new file mode 100644 index 000000000000..7415a85c4998 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con1_ccb-NIXL.yaml @@ -0,0 +1,107 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1' + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 3 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 16 + max_num_tokens: 64 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con1_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con1_ccb-UCX.yaml new file mode 100644 index 000000000000..aac703968ade --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con1_ccb-UCX.yaml @@ -0,0 +1,107 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1' + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 3 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 16 + max_num_tokens: 64 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con2_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con2_ccb-NIXL.yaml new file mode 100644 index 000000000000..5aabc9772ea2 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con2_ccb-NIXL.yaml @@ -0,0 +1,107 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '2' + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 3 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 16 + max_num_tokens: 64 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con2_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con2_ccb-UCX.yaml new file mode 100644 index 000000000000..b3f644b5ae4a --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con2_ccb-UCX.yaml @@ -0,0 +1,107 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '2' + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 3 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 16 + max_num_tokens: 64 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con4_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con4_ccb-NIXL.yaml new file mode 100644 index 000000000000..0a48fac6f35a --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con4_ccb-NIXL.yaml @@ -0,0 +1,107 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '4' + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 3 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 16 + max_num_tokens: 64 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con4_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con4_ccb-UCX.yaml new file mode 100644 index 000000000000..bbfe945b8f32 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con4_ccb-UCX.yaml @@ -0,0 +1,107 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '4' + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 3 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 16 + max_num_tokens: 64 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con8_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con8_ccb-NIXL.yaml new file mode 100644 index 000000000000..c59b6d28e6bd --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con8_ccb-NIXL.yaml @@ -0,0 +1,107 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '8' + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 3 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 16 + max_num_tokens: 64 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con8_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con8_ccb-UCX.yaml new file mode 100644 index 000000000000..47037ef0c180 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs16_eplb0_mtp3_con8_ccb-UCX.yaml @@ -0,0 +1,107 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '8' + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 3 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 16 + max_num_tokens: 64 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con16_ccb-NIXL.yaml similarity index 89% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con16_ccb-NIXL.yaml index 8a0bc5ca82ea..ef0890f0e011 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con16_ccb-NIXL.yaml @@ -13,7 +13,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: e2e @@ -21,7 +21,7 @@ benchmark: multi_round: 8 benchmark_ratio: 0.8 streaming: true - concurrency_list: 1 2 4 8 16 32 + concurrency_list: '16' input_length: 8192 output_length: 1024 dataset_file: @@ -36,8 +36,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con16_ccb-UCX.yaml similarity index 89% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con16_ccb-UCX.yaml index 3328a559c3e3..831c08f2516c 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con16_ccb-UCX.yaml @@ -13,7 +13,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: e2e @@ -21,7 +21,7 @@ benchmark: multi_round: 8 benchmark_ratio: 0.8 streaming: true - concurrency_list: 1 2 4 8 16 32 + concurrency_list: '16' input_length: 8192 output_length: 1024 dataset_file: @@ -36,8 +36,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL.yaml new file mode 100644 index 000000000000..9af25336a776 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL.yaml @@ -0,0 +1,101 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1' + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 3 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 32 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con1_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con1_ccb-UCX.yaml new file mode 100644 index 000000000000..6fe180ca0534 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con1_ccb-UCX.yaml @@ -0,0 +1,101 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '1' + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 3 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 32 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL.yaml new file mode 100644 index 000000000000..c5d27e0e4bfa --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL.yaml @@ -0,0 +1,101 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '2' + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 3 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 32 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con2_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con2_ccb-UCX.yaml new file mode 100644 index 000000000000..69b6e98e389e --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con2_ccb-UCX.yaml @@ -0,0 +1,101 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '2' + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 3 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 32 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con32_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con32_ccb-NIXL.yaml new file mode 100644 index 000000000000..8f9aecb808ef --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con32_ccb-NIXL.yaml @@ -0,0 +1,101 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '32' + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 3 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 32 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con32_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con32_ccb-UCX.yaml new file mode 100644 index 000000000000..48dceec3c08b --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con32_ccb-UCX.yaml @@ -0,0 +1,101 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '32' + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 3 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 32 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL.yaml new file mode 100644 index 000000000000..1d16b8317398 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL.yaml @@ -0,0 +1,101 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '4' + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 3 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 32 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con4_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con4_ccb-UCX.yaml new file mode 100644 index 000000000000..c15166fc1191 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con4_ccb-UCX.yaml @@ -0,0 +1,101 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '4' + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 3 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 32 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL.yaml new file mode 100644 index 000000000000..25f9e4045a5a --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL.yaml @@ -0,0 +1,101 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '8' + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 3 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 32 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: NIXL diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con8_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con8_ccb-UCX.yaml new file mode 100644 index 000000000000..93f024b65fc2 --- /dev/null +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx1_gen3_tep8_bs32_eplb0_mtp0_con8_ccb-UCX.yaml @@ -0,0 +1,101 @@ +metadata: + model_name: deepseek_r1_0528_fp4_v2 + precision: fp4 + model_dir_name: DeepSeek-R1-0528-FP4-v2 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 8k1k +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: e2e + use_nv_sa_benchmark: true + multi_round: 8 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '8' + input_length: 8192 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 3 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + pipeline_parallel_size: 1 + max_batch_size: 32 + max_num_tokens: 32 + max_seq_len: 9419 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: TRTLLM + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + allreduce_strategy: MNNVL + ctx: + max_batch_size: 1 + max_num_tokens: 8448 + max_seq_len: 9419 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb0_mtp0_con1024_ccb-NIXL.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb0_mtp0_ccb-NIXL.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb0_mtp0_con1024_ccb-NIXL.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb0_mtp0_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb0_mtp0_con1024_ccb-UCX.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb0_mtp0_ccb-UCX.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb0_mtp0_con1024_ccb-UCX.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb0_mtp3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb0_mtp3_con512_ccb-NIXL.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb0_mtp3_ccb-NIXL.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb0_mtp3_con512_ccb-NIXL.yaml diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb0_mtp3_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb0_mtp3_con512_ccb-UCX.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb0_mtp3_ccb-UCX.yaml rename to tests/scripts/perf/disaggregated/deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb0_mtp3_con512_ccb-UCX.yaml diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-NIXL.yaml similarity index 91% rename from tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-NIXL.yaml rename to tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-NIXL.yaml index bcbf7532af69..74828e6d9a36 100644 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-NIXL.yaml @@ -14,7 +14,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: gen_only @@ -22,7 +22,7 @@ benchmark: multi_round: 1 benchmark_ratio: 0.8 streaming: true - concurrency_list: 512 1024 + concurrency_list: '1024' input_length: 1024 output_length: 1024 dataset_file: @@ -37,8 +37,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-UCX.yaml similarity index 91% rename from tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-UCX.yaml rename to tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-UCX.yaml index 579ed9445bec..9630d2c40a40 100644 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-UCX.yaml @@ -14,7 +14,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: gen_only @@ -22,7 +22,7 @@ benchmark: multi_round: 1 benchmark_ratio: 0.8 streaming: true - concurrency_list: 512 1024 + concurrency_list: '1024' input_length: 1024 output_length: 1024 dataset_file: @@ -37,8 +37,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-NIXL.yaml new file mode 100644 index 000000000000..2e5291955167 --- /dev/null +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-NIXL.yaml @@ -0,0 +1,114 @@ +metadata: + model_name: Qwen3-235B-A22B-FP4 + precision: fp4 + model_dir_name: Qwen3-235B-A22B-FP4 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k + dataset_file: disagg_datasets/deepseek-r1-1024-1024-100000-ratio-1_for_serve.json +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: gen_only + use_nv_sa_benchmark: false + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '512' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + enable_layerwise_nvtx_marker: true + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + pipeline_parallel_size: 1 + max_batch_size: 64 + max_num_tokens: 256 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: WIDEEP + use_low_precision_moe_combine: true + load_balancer: + num_slots: 288 + layer_updates_per_iter: 1 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + enable_layerwise_nvtx_marker: true + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: NIXL + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-UCX.yaml new file mode 100644 index 000000000000..248b13a9a8c1 --- /dev/null +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-UCX.yaml @@ -0,0 +1,114 @@ +metadata: + model_name: Qwen3-235B-A22B-FP4 + precision: fp4 + model_dir_name: Qwen3-235B-A22B-FP4 + supported_gpus: + - GB200 + - GB300 + script_file: disaggr_torch.slurm + benchmark_type: 1k1k + dataset_file: disagg_datasets/deepseek-r1-1024-1024-100000-ratio-1_for_serve.json +slurm: + script_file: disaggr_torch.slurm + partition: + account: + job_time: 02:00:00 + job_name: unified-benchmark + extra_args: --gres=gpu:4 + numa_bind: true +benchmark: + mode: gen_only + use_nv_sa_benchmark: false + multi_round: 1 + benchmark_ratio: 0.8 + streaming: true + concurrency_list: '512' + input_length: 1024 + output_length: 1024 + dataset_file: +hardware: + gpus_per_node: 4 + num_ctx_servers: 1 + num_gen_servers: 1 +environment: + container_mount: + container_image: + model_path: + trtllm_repo: '' + build_wheel: false + work_dir: + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 +profiling: + nsys_on: false +accuracy: + enable_accuracy_test: false +worker_config: + gen: + enable_layerwise_nvtx_marker: true + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + pipeline_parallel_size: 1 + max_batch_size: 64 + max_num_tokens: 256 + max_seq_len: 2251 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 32 + - 64 + - 128 + - 256 + - 512 + - 768 + - 1024 + - 2048 + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + dtype: fp8 + moe_config: + backend: WIDEEP + use_low_precision_moe_combine: true + load_balancer: + num_slots: 288 + layer_updates_per_iter: 1 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + stream_interval: 20 + num_postprocess_workers: 4 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + ctx: + enable_layerwise_nvtx_marker: true + max_batch_size: 4 + max_num_tokens: 4608 + max_seq_len: 2251 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 4608 + backend: UCX + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml rename to tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_ccb-UCX.yaml rename to tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-NIXL.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_ccb-NIXL.yaml rename to tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-NIXL.yaml diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-UCX.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_ccb-UCX.yaml rename to tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-UCX.yaml diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL.yaml rename to tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL.yaml diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL_kv-reuse.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL_kv-reuse.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL_kv-reuse.yaml rename to tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL_kv-reuse.yaml diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-UCX.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-UCX.yaml rename to tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-UCX.yaml diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-NIXL.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-NIXL.yaml rename to tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-NIXL.yaml diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-UCX.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-UCX.yaml rename to tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-UCX.yaml diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_con12288_ccb-DEFAULT.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT.yaml rename to tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_con12288_ccb-DEFAULT.yaml diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_con1024_ccb-DEFAULT.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT.yaml rename to tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_con1024_ccb-DEFAULT.yaml diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-NIXL.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL.yaml rename to tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-NIXL.yaml diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-UCX.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-UCX.yaml rename to tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-UCX.yaml diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml rename to tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-UCX.yaml rename to tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_ccb-NIXL.yaml rename to tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL.yaml diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-NIXL.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_ccb-NIXL.yaml rename to tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-NIXL.yaml diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_con12288_ccb-DEFAULT.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_ccb-DEFAULT.yaml rename to tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_con12288_ccb-DEFAULT.yaml diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_con1024_ccb-DEFAULT.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_ccb-DEFAULT.yaml rename to tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_con1024_ccb-DEFAULT.yaml diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-NIXL.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_ccb-NIXL.yaml rename to tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-NIXL.yaml diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_ccb-NIXL.yaml rename to tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml diff --git a/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_1k1k_ctx3_gen1_dep32_bs1024_eplb384_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_1k1k_ctx3_gen1_dep32_bs1024_eplb384_mtp0_con16384_ccb-NIXL.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_1k1k_ctx3_gen1_dep32_bs1024_eplb384_mtp0_ccb-NIXL.yaml rename to tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_1k1k_ctx3_gen1_dep32_bs1024_eplb384_mtp0_con16384_ccb-NIXL.yaml diff --git a/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_8k1k_ctx8_gen1_dep32_bs256_eplb416_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_8k1k_ctx8_gen1_dep32_bs256_eplb416_mtp0_con8192_ccb-NIXL.yaml similarity index 100% rename from tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_8k1k_ctx8_gen1_dep32_bs256_eplb416_mtp0_ccb-NIXL.yaml rename to tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_8k1k_ctx8_gen1_dep32_bs256_eplb416_mtp0_con8192_ccb-NIXL.yaml From 6a2306b317d4768dcca5dfebff2cda3f5c99db2f Mon Sep 17 00:00:00 2001 From: FredricZ-2007 <226039983+fredricz-20070104@users.noreply.github.com> Date: Fri, 6 Mar 2026 07:41:32 +0000 Subject: [PATCH 07/12] update model name to adapt current model path in test perf sanity Signed-off-by: FredricZ-2007 <226039983+fredricz-20070104@users.noreply.github.com> --- ...128k8k_ctx1_pp4_gen13_tep4_bs1_eplb0_mtp0_con1-Default.yaml | 2 +- ..._128k8k_ctx1_pp4_gen5_tep4_bs4_eplb0_mtp0_con4-Default.yaml | 2 +- ..._128k8k_ctx1_pp4_gen6_tep8_bs1_eplb0_mtp3_con1-Default.yaml | 3 +-- ..._128k8k_ctx1_pp4_gen7_tep8_bs1_eplb0_mtp0_con1-Default.yaml | 2 +- ..._128k8k_ctx1_pp4_gen8_tep4_bs2_eplb0_mtp0_con2-Default.yaml | 2 +- ..._128k8k_ctx1_pp4_gen8_tep8_bs1_eplb0_mtp0_con1-Default.yaml | 2 +- ...128k8k_ctx1_pp8_gen11_tep4_bs2_eplb0_mtp0_con2-Default.yaml | 2 +- ...128k8k_ctx1_pp8_gen14_tep4_bs1_eplb0_mtp0_con1-Default.yaml | 2 +- ...128k8k_ctx1_pp8_gen1_dep16_bs1_eplb0_mtp3_con1-Default.yaml | 3 +-- ..._128k8k_ctx1_pp8_gen1_dep8_bs4_eplb0_mtp2_con4-Default.yaml | 3 +-- ..._128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp0_con1-Default.yaml | 2 +- ..._128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp3_con1-Default.yaml | 3 +-- ..._128k8k_ctx1_pp8_gen1_tep8_bs2_eplb0_mtp3_con2-Default.yaml | 3 +-- ..._128k8k_ctx1_pp8_gen5_tep8_bs2_eplb0_mtp3_con2-Default.yaml | 2 +- ..._128k8k_ctx1_pp8_gen7_tep4_bs2_eplb0_mtp2_con2-Default.yaml | 2 +- ..._128k8k_ctx1_pp8_gen7_tep8_bs1_eplb0_mtp0_con1-Default.yaml | 2 +- ..._128k8k_ctx1_pp8_gen8_tep4_bs4_eplb0_mtp0_con4-Default.yaml | 2 +- ..._128k8k_ctx2_pp4_gen7_tep8_bs2_eplb0_mtp3_con2-Default.yaml | 2 +- ...128k8k_ctx2_pp8_gen1_dep16_bs8_eplb0_mtp0_con8-Default.yaml | 2 +- ...128k8k_ctx2_pp8_gen1_dep32_bs2_eplb0_mtp0_con2-Default.yaml | 2 +- ...8k8k_ctx3_pp4_gen1_dep8_bs16_eplb0_mtp1_con128-Default.yaml | 2 +- ...8k8k_ctx3_pp8_gen1_dep16_bs16_eplb0_mtp0_con16-Default.yaml | 2 +- ...128k8k_ctx3_pp8_gen1_dep16_bs8_eplb0_mtp2_con8-Default.yaml | 3 +-- ...128k8k_ctx3_pp8_gen1_dep32_bs2_eplb0_mtp3_con2-Default.yaml | 3 +-- ...128k8k_ctx3_pp8_gen1_dep32_bs4_eplb0_mtp0_con4-Default.yaml | 2 +- ...k8k_ctx5_pp4_gen1_dep16_bs16_eplb0_mtp0_con256-Default.yaml | 2 +- ...8k8k_ctx5_pp4_gen1_dep16_bs8_eplb0_mtp3_con128-Default.yaml | 3 +-- ...28k8k_ctx5_pp4_gen1_dep32_bs2_eplb0_mtp3_con64-Default.yaml | 3 +-- ...8k8k_ctx5_pp4_gen1_dep32_bs4_eplb0_mtp0_con128-Default.yaml | 2 +- ...k8k_ctx7_pp4_gen1_dep16_bs16_eplb0_mtp1_con256-Default.yaml | 3 +-- ...k8k_ctx7_pp4_gen1_dep16_bs32_eplb0_mtp0_con512-Default.yaml | 2 +- ...k8k_ctx8_pp4_gen1_dep16_bs32_eplb0_mtp1_con512-Default.yaml | 3 +-- ...8k8k_ctx8_pp4_gen1_dep32_bs4_eplb0_mtp3_con128-Default.yaml | 3 +-- ...8k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp0_con256-Default.yaml | 2 +- ...8k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp3_con256-Default.yaml | 3 +-- 35 files changed, 35 insertions(+), 48 deletions(-) diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen13_tep4_bs1_eplb0_mtp0_con1-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen13_tep4_bs1_eplb0_mtp0_con1-Default.yaml index a4ad607842d9..3887effd1e1e 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen13_tep4_bs1_eplb0_mtp0_con1-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen13_tep4_bs1_eplb0_mtp0_con1-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen5_tep4_bs4_eplb0_mtp0_con4-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen5_tep4_bs4_eplb0_mtp0_con4-Default.yaml index f2b1074ea4fc..70b4dd94a61f 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen5_tep4_bs4_eplb0_mtp0_con4-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen5_tep4_bs4_eplb0_mtp0_con4-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen6_tep8_bs1_eplb0_mtp3_con1-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen6_tep8_bs1_eplb0_mtp3_con1-Default.yaml index 3f9ef0ebaa12..327c0ae797df 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen6_tep8_bs1_eplb0_mtp3_con1-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen6_tep8_bs1_eplb0_mtp3_con1-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: @@ -49,7 +49,6 @@ worker_config: enable_attention_dp: false pipeline_parallel_size: 1 max_batch_size: 1 - # mtp_size=3 ⇒ max_num_tokens = 1 * (3 + 1) = 4 max_num_tokens: 4 max_seq_len: 139296 cuda_graph_config: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen7_tep8_bs1_eplb0_mtp0_con1-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen7_tep8_bs1_eplb0_mtp0_con1-Default.yaml index ca20d690388a..e966f0280eb4 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen7_tep8_bs1_eplb0_mtp0_con1-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen7_tep8_bs1_eplb0_mtp0_con1-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep4_bs2_eplb0_mtp0_con2-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep4_bs2_eplb0_mtp0_con2-Default.yaml index a22245dee91a..61bee8027a91 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep4_bs2_eplb0_mtp0_con2-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep4_bs2_eplb0_mtp0_con2-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep8_bs1_eplb0_mtp0_con1-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep8_bs1_eplb0_mtp0_con1-Default.yaml index 59d835780b16..1ae024973614 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep8_bs1_eplb0_mtp0_con1-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp4_gen8_tep8_bs1_eplb0_mtp0_con1-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen11_tep4_bs2_eplb0_mtp0_con2-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen11_tep4_bs2_eplb0_mtp0_con2-Default.yaml index 5be853b4ac63..7b3066f28687 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen11_tep4_bs2_eplb0_mtp0_con2-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen11_tep4_bs2_eplb0_mtp0_con2-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen14_tep4_bs1_eplb0_mtp0_con1-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen14_tep4_bs1_eplb0_mtp0_con1-Default.yaml index dfbf6222ed95..a0e0717b0e33 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen14_tep4_bs1_eplb0_mtp0_con1-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen14_tep4_bs1_eplb0_mtp0_con1-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep16_bs1_eplb0_mtp3_con1-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep16_bs1_eplb0_mtp3_con1-Default.yaml index 55972145a59f..c382932be0b4 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep16_bs1_eplb0_mtp3_con1-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep16_bs1_eplb0_mtp3_con1-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: @@ -49,7 +49,6 @@ worker_config: enable_attention_dp: true pipeline_parallel_size: 1 max_batch_size: 1 - # mtp_size=3 ⇒ max_num_tokens = 1 * (3 + 1) = 4 max_num_tokens: 4 max_seq_len: 139296 cuda_graph_config: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep8_bs4_eplb0_mtp2_con4-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep8_bs4_eplb0_mtp2_con4-Default.yaml index 02bc0863a9a0..3f4aee7d7532 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep8_bs4_eplb0_mtp2_con4-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_dep8_bs4_eplb0_mtp2_con4-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: @@ -49,7 +49,6 @@ worker_config: enable_attention_dp: true pipeline_parallel_size: 1 max_batch_size: 4 - # mtp_size=2 ⇒ max_num_tokens = 4 * (2 + 1) = 12 max_num_tokens: 12 max_seq_len: 139296 cuda_graph_config: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp0_con1-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp0_con1-Default.yaml index fd3ad8c0f1be..7f4394f574bc 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp0_con1-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp0_con1-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp3_con1-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp3_con1-Default.yaml index 1cd58fc3936e..a5de96fc5053 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp3_con1-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs1_eplb0_mtp3_con1-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: @@ -49,7 +49,6 @@ worker_config: enable_attention_dp: false pipeline_parallel_size: 1 max_batch_size: 1 - # mtp_size=3 ⇒ max_num_tokens = 1 * (3 + 1) = 4 max_num_tokens: 4 max_seq_len: 139296 cuda_graph_config: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs2_eplb0_mtp3_con2-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs2_eplb0_mtp3_con2-Default.yaml index 1565c88347c7..d447a67b3a51 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs2_eplb0_mtp3_con2-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen1_tep8_bs2_eplb0_mtp3_con2-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: @@ -49,7 +49,6 @@ worker_config: enable_attention_dp: false pipeline_parallel_size: 1 max_batch_size: 2 - # mtp_size=3 ⇒ max_num_tokens = 2 * (3 + 1) = 8 max_num_tokens: 8 max_seq_len: 139296 cuda_graph_config: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen5_tep8_bs2_eplb0_mtp3_con2-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen5_tep8_bs2_eplb0_mtp3_con2-Default.yaml index 281ab8215104..d45c42bc8ccc 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen5_tep8_bs2_eplb0_mtp3_con2-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen5_tep8_bs2_eplb0_mtp3_con2-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep4_bs2_eplb0_mtp2_con2-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep4_bs2_eplb0_mtp2_con2-Default.yaml index 77e113fec290..0c8a38335489 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep4_bs2_eplb0_mtp2_con2-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep4_bs2_eplb0_mtp2_con2-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep8_bs1_eplb0_mtp0_con1-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep8_bs1_eplb0_mtp0_con1-Default.yaml index 517b5c61e778..9a6391945094 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep8_bs1_eplb0_mtp0_con1-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen7_tep8_bs1_eplb0_mtp0_con1-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen8_tep4_bs4_eplb0_mtp0_con4-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen8_tep4_bs4_eplb0_mtp0_con4-Default.yaml index 449fd368a3a8..d0274f8a00db 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen8_tep4_bs4_eplb0_mtp0_con4-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx1_pp8_gen8_tep4_bs4_eplb0_mtp0_con4-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp4_gen7_tep8_bs2_eplb0_mtp3_con2-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp4_gen7_tep8_bs2_eplb0_mtp3_con2-Default.yaml index d794643060ab..9167382bddff 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp4_gen7_tep8_bs2_eplb0_mtp3_con2-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp4_gen7_tep8_bs2_eplb0_mtp3_con2-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep16_bs8_eplb0_mtp0_con8-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep16_bs8_eplb0_mtp0_con8-Default.yaml index ff9a9e62cf8b..2d6760bf7dca 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep16_bs8_eplb0_mtp0_con8-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep16_bs8_eplb0_mtp0_con8-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep32_bs2_eplb0_mtp0_con2-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep32_bs2_eplb0_mtp0_con2-Default.yaml index d547dae70666..f88d250e52b8 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep32_bs2_eplb0_mtp0_con2-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx2_pp8_gen1_dep32_bs2_eplb0_mtp0_con2-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp4_gen1_dep8_bs16_eplb0_mtp1_con128-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp4_gen1_dep8_bs16_eplb0_mtp1_con128-Default.yaml index 90d277005791..37abd60ca4d6 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp4_gen1_dep8_bs16_eplb0_mtp1_con128-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp4_gen1_dep8_bs16_eplb0_mtp1_con128-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs16_eplb0_mtp0_con16-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs16_eplb0_mtp0_con16-Default.yaml index 1c935ff7c416..458b34c4dcc9 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs16_eplb0_mtp0_con16-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs16_eplb0_mtp0_con16-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs8_eplb0_mtp2_con8-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs8_eplb0_mtp2_con8-Default.yaml index ee9d98cdaa99..3cef4212ced2 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs8_eplb0_mtp2_con8-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep16_bs8_eplb0_mtp2_con8-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: @@ -49,7 +49,6 @@ worker_config: enable_attention_dp: true pipeline_parallel_size: 1 max_batch_size: 8 - # mtp_size=2 ⇒ max_num_tokens = 8 * (2 + 1) = 24 max_num_tokens: 24 max_seq_len: 139296 cuda_graph_config: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs2_eplb0_mtp3_con2-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs2_eplb0_mtp3_con2-Default.yaml index d69db0a1ca34..b921750a006c 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs2_eplb0_mtp3_con2-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs2_eplb0_mtp3_con2-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: @@ -49,7 +49,6 @@ worker_config: enable_attention_dp: true pipeline_parallel_size: 1 max_batch_size: 2 - # mtp_size=3 ⇒ max_num_tokens = 2 * (3 + 1) = 8 max_num_tokens: 8 max_seq_len: 139296 cuda_graph_config: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs4_eplb0_mtp0_con4-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs4_eplb0_mtp0_con4-Default.yaml index a51d1073e30c..85967f76c162 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs4_eplb0_mtp0_con4-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx3_pp8_gen1_dep32_bs4_eplb0_mtp0_con4-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs16_eplb0_mtp0_con256-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs16_eplb0_mtp0_con256-Default.yaml index 05d6a10d3268..ea7c356b6118 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs16_eplb0_mtp0_con256-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs16_eplb0_mtp0_con256-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs8_eplb0_mtp3_con128-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs8_eplb0_mtp3_con128-Default.yaml index 5befdee83318..ac17975c3bc4 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs8_eplb0_mtp3_con128-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep16_bs8_eplb0_mtp3_con128-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: @@ -49,7 +49,6 @@ worker_config: enable_attention_dp: true pipeline_parallel_size: 1 max_batch_size: 8 - # mtp_size=3 ⇒ max_num_tokens = 8 * (3 + 1) = 32 max_num_tokens: 32 max_seq_len: 139296 cuda_graph_config: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs2_eplb0_mtp3_con64-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs2_eplb0_mtp3_con64-Default.yaml index e2bcac62240b..9d2a844f5d79 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs2_eplb0_mtp3_con64-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs2_eplb0_mtp3_con64-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: @@ -49,7 +49,6 @@ worker_config: enable_attention_dp: true pipeline_parallel_size: 1 max_batch_size: 2 - # mtp_size=3 ⇒ max_num_tokens = 2 * (3 + 1) = 8 max_num_tokens: 8 max_seq_len: 139296 cuda_graph_config: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs4_eplb0_mtp0_con128-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs4_eplb0_mtp0_con128-Default.yaml index d449c173c778..37b38f9c1af1 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs4_eplb0_mtp0_con128-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx5_pp4_gen1_dep32_bs4_eplb0_mtp0_con128-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs16_eplb0_mtp1_con256-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs16_eplb0_mtp1_con256-Default.yaml index 90ed3bd0d33c..c7489d2864d0 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs16_eplb0_mtp1_con256-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs16_eplb0_mtp1_con256-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: @@ -49,7 +49,6 @@ worker_config: enable_attention_dp: true pipeline_parallel_size: 1 max_batch_size: 16 - # mtp_size=1 ⇒ max_num_tokens = 16 * (1 + 1) = 32 max_num_tokens: 32 max_seq_len: 139296 cuda_graph_config: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs32_eplb0_mtp0_con512-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs32_eplb0_mtp0_con512-Default.yaml index 2eed9c9959c7..3492ea65c42d 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs32_eplb0_mtp0_con512-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx7_pp4_gen1_dep16_bs32_eplb0_mtp0_con512-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep16_bs32_eplb0_mtp1_con512-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep16_bs32_eplb0_mtp1_con512-Default.yaml index c9226167aa5a..7be406bdc39d 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep16_bs32_eplb0_mtp1_con512-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep16_bs32_eplb0_mtp1_con512-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: @@ -49,7 +49,6 @@ worker_config: enable_attention_dp: true pipeline_parallel_size: 1 max_batch_size: 32 - # mtp_size=1 ⇒ max_num_tokens = 32 * (1 + 1) = 64 max_num_tokens: 64 max_seq_len: 139296 cuda_graph_config: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs4_eplb0_mtp3_con128-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs4_eplb0_mtp3_con128-Default.yaml index e92e50d77f95..7a34ac9edd0d 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs4_eplb0_mtp3_con128-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs4_eplb0_mtp3_con128-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: @@ -49,7 +49,6 @@ worker_config: enable_attention_dp: true pipeline_parallel_size: 1 max_batch_size: 4 - # mtp_size=3 ⇒ max_num_tokens = 4 * (3 + 1) = 16 max_num_tokens: 16 max_seq_len: 139296 cuda_graph_config: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp0_con256-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp0_con256-Default.yaml index fb0c3d54835c..9402160635d1 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp0_con256-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp0_con256-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp3_con256-Default.yaml b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp3_con256-Default.yaml index 8740d2861c4f..c19dc3ed6f0e 100644 --- a/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp3_con256-Default.yaml +++ b/tests/scripts/perf/disaggregated/deepseek-r1-fp4_128k8k_ctx8_pp4_gen1_dep32_bs8_eplb0_mtp3_con256-Default.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: @@ -49,7 +49,6 @@ worker_config: enable_attention_dp: true pipeline_parallel_size: 1 max_batch_size: 8 - # mtp_size=3 ⇒ max_num_tokens = 8 * (3 + 1) = 32 max_num_tokens: 32 max_seq_len: 139296 cuda_graph_config: From 2f0d5171755372cd180711461f37e6daae86af44 Mon Sep 17 00:00:00 2001 From: FredricZ-2007 <226039983+fredricz-20070104@users.noreply.github.com> Date: Fri, 6 Mar 2026 07:57:23 +0000 Subject: [PATCH 08/12] update model names Signed-off-by: FredricZ-2007 <226039983+fredricz-20070104@users.noreply.github.com> --- tests/integration/defs/perf/test_perf_sanity.py | 1 + ...1_gen1_dep16_bs64_eplb0_mtp3_con1024_ccb-NIXL.yaml | 2 +- ...x1_gen1_dep16_bs64_eplb0_mtp3_con1024_ccb-UCX.yaml | 2 +- ...x1_gen1_dep16_bs64_eplb0_mtp3_con512_ccb-NIXL.yaml | 2 +- ...tx1_gen1_dep16_bs64_eplb0_mtp3_con512_ccb-UCX.yaml | 2 +- ...x1_gen1_dep32_bs16_eplb0_mtp3_con512_ccb-NIXL.yaml | 9 +++++---- ...tx1_gen1_dep32_bs16_eplb0_mtp3_con512_ccb-UCX.yaml | 9 +++++---- ...ctx1_gen4_tep8_bs32_eplb0_mtp0_con16_ccb-NIXL.yaml | 2 +- ..._ctx1_gen4_tep8_bs32_eplb0_mtp0_con16_ccb-UCX.yaml | 2 +- ..._ctx1_gen4_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL.yaml | 2 +- ...k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con1_ccb-UCX.yaml | 2 +- ..._ctx1_gen4_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL.yaml | 2 +- ...k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con2_ccb-UCX.yaml | 2 +- ...ctx1_gen4_tep8_bs32_eplb0_mtp0_con32_ccb-NIXL.yaml | 2 +- ..._ctx1_gen4_tep8_bs32_eplb0_mtp0_con32_ccb-UCX.yaml | 2 +- ..._ctx1_gen4_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL.yaml | 2 +- ...k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con4_ccb-UCX.yaml | 2 +- ..._ctx1_gen4_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL.yaml | 2 +- ...k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con8_ccb-UCX.yaml | 2 +- ..._gen1_dep16_bs128_eplb0_mtp1_con2048_ccb-NIXL.yaml | 9 +++++---- ...2_gen1_dep16_bs128_eplb0_mtp1_con2048_ccb-UCX.yaml | 9 +++++---- ...ctx1_gen1_tep8_bs32_eplb0_mtp0_con16_ccb-NIXL.yaml | 2 +- ..._ctx1_gen1_tep8_bs32_eplb0_mtp0_con16_ccb-UCX.yaml | 2 +- ..._ctx1_gen1_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL.yaml | 2 +- ...k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con1_ccb-UCX.yaml | 2 +- ..._ctx1_gen1_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL.yaml | 2 +- ...k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con2_ccb-UCX.yaml | 2 +- ...ctx1_gen1_tep8_bs32_eplb0_mtp0_con36_ccb-NIXL.yaml | 2 +- ..._ctx1_gen1_tep8_bs32_eplb0_mtp0_con36_ccb-UCX.yaml | 2 +- ..._ctx1_gen1_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL.yaml | 2 +- ...k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con4_ccb-UCX.yaml | 2 +- ..._ctx1_gen1_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL.yaml | 2 +- ...k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con8_ccb-UCX.yaml | 2 +- ...gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-NIXL.yaml | 2 +- ..._gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-UCX.yaml | 2 +- ..._gen1_dep16_bs64_eplb288_mtp3_con512_ccb-NIXL.yaml | 2 +- ...1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-UCX.yaml | 2 +- ..._gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml | 9 +++++---- ...1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml | 9 +++++---- ...en1_dep16_bs128_eplb288_mtp1_con2048_ccb-NIXL.yaml | 9 +++++---- ...gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-UCX.yaml | 9 +++++---- ...gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL.yaml | 9 +++++---- ...2_bs32_eplb288_mtp0_con1024_ccb-NIXL_kv-reuse.yaml | 9 +++++---- ..._gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-UCX.yaml | 9 +++++---- ...en1_dep16_bs128_eplb288_mtp3_con2048_ccb-NIXL.yaml | 9 +++++---- ...gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-UCX.yaml | 9 +++++---- ..._dep48_bs16_eplb288_mtp3_con12288_ccb-DEFAULT.yaml | 10 +++++----- ..._dep32_bs128_eplb288_mtp3_con1024_ccb-DEFAULT.yaml | 9 +++++---- ...gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-NIXL.yaml | 9 +++++---- ..._gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-UCX.yaml | 9 +++++---- ..._gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml | 9 +++++---- ...8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml | 9 +++++---- ...gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL.yaml | 11 ++++++----- ...en1_dep16_bs128_eplb288_mtp3_con2048_ccb-NIXL.yaml | 9 +++++---- ..._dep48_bs16_eplb288_mtp3_con12288_ccb-DEFAULT.yaml | 10 +++++----- ..._dep32_bs128_eplb288_mtp3_con1024_ccb-DEFAULT.yaml | 9 +++++---- ...gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-NIXL.yaml | 9 +++++---- ..._gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml | 9 +++++---- ...1_dep32_bs1024_eplb384_mtp0_con16384_ccb-NIXL.yaml | 4 ++-- ...en1_dep32_bs256_eplb416_mtp0_con8192_ccb-NIXL.yaml | 4 ++-- 60 files changed, 163 insertions(+), 139 deletions(-) diff --git a/tests/integration/defs/perf/test_perf_sanity.py b/tests/integration/defs/perf/test_perf_sanity.py index 58e316d7f2fc..66b7e9a41843 100644 --- a/tests/integration/defs/perf/test_perf_sanity.py +++ b/tests/integration/defs/perf/test_perf_sanity.py @@ -55,6 +55,7 @@ "gpt_oss_120b_fp4": "gpt_oss/gpt-oss-120b", "k2_thinking_fp4": "Kimi-K2-Thinking-NVFP4", "qwen3_235b_a22b_fp4": "Qwen3/saved_models_Qwen3-235B-A22B_nvfp4_hf", # Qwen3-235B-A22B-FP4 + "qwen3_235b_a22b_fp8": "Qwen3/saved_models_Qwen3-235B-A22B_fp8_hf", # Qwen3-235B-A22B-FP8 } SUPPORTED_GPU_MAPPING = { diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con1024_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con1024_ccb-NIXL.yaml index 1398cf8cc87e..d9d0ae88cdb7 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con1024_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con1024_ccb-NIXL.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP4 + model_name: qwen3_235b_a22b_fp4 precision: fp4 model_dir_name: Qwen3-235B-A22B-FP4 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con1024_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con1024_ccb-UCX.yaml index ac5fceb71884..6a1796bf8301 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con1024_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con1024_ccb-UCX.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP4 + model_name: qwen3_235b_a22b_fp4 precision: fp4 model_dir_name: Qwen3-235B-A22B-FP4 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con512_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con512_ccb-NIXL.yaml index fa99044fce7b..ab5bd95a7ba1 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con512_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con512_ccb-NIXL.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP4 + model_name: qwen3_235b_a22b_fp4 precision: fp4 model_dir_name: Qwen3-235B-A22B-FP4 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con512_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con512_ccb-UCX.yaml index e8475a5df2f4..02540e48a286 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con512_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb0_mtp3_con512_ccb-UCX.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP4 + model_name: qwen3_235b_a22b_fp4 precision: fp4 model_dir_name: Qwen3-235B-A22B-FP4 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_con512_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_con512_ccb-NIXL.yaml index 213c6ced5f33..1183e9719258 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_con512_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_con512_ccb-NIXL.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP4 + model_name: qwen3_235b_a22b_fp4 precision: fp4 model_dir_name: Qwen3-235B-A22B-FP4 supported_gpus: @@ -13,7 +13,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: e2e @@ -36,8 +36,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_con512_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_con512_ccb-UCX.yaml index d7279d3b219a..38bd0ddc4e94 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_con512_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb0_mtp3_con512_ccb-UCX.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP4 + model_name: qwen3_235b_a22b_fp4 precision: fp4 model_dir_name: Qwen3-235B-A22B-FP4 supported_gpus: @@ -13,7 +13,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: e2e @@ -36,8 +36,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con16_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con16_ccb-NIXL.yaml index 72ff6a3f579a..a63bb52ec969 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con16_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con16_ccb-NIXL.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP4 + model_name: qwen3_235b_a22b_fp4 precision: fp4 model_dir_name: Qwen3-235B-A22B-FP4 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con16_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con16_ccb-UCX.yaml index c158ae77ba08..573c0b94cdbf 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con16_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con16_ccb-UCX.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP4 + model_name: qwen3_235b_a22b_fp4 precision: fp4 model_dir_name: Qwen3-235B-A22B-FP4 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL.yaml index 7c77cdf5e035..f8b623a0cadf 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP4 + model_name: qwen3_235b_a22b_fp4 precision: fp4 model_dir_name: Qwen3-235B-A22B-FP4 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con1_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con1_ccb-UCX.yaml index d978c98e1922..13f48032da49 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con1_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con1_ccb-UCX.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP4 + model_name: qwen3_235b_a22b_fp4 precision: fp4 model_dir_name: Qwen3-235B-A22B-FP4 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL.yaml index 224d3b9b659d..c59e682d51d5 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP4 + model_name: qwen3_235b_a22b_fp4 precision: fp4 model_dir_name: Qwen3-235B-A22B-FP4 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con2_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con2_ccb-UCX.yaml index 347b8645b1c0..b3ec30d4232a 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con2_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con2_ccb-UCX.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP4 + model_name: qwen3_235b_a22b_fp4 precision: fp4 model_dir_name: Qwen3-235B-A22B-FP4 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con32_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con32_ccb-NIXL.yaml index 1a92f6c81925..29d4120b302c 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con32_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con32_ccb-NIXL.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP4 + model_name: qwen3_235b_a22b_fp4 precision: fp4 model_dir_name: Qwen3-235B-A22B-FP4 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con32_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con32_ccb-UCX.yaml index 8d09c7ec0d14..99cfeacc8929 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con32_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con32_ccb-UCX.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP4 + model_name: qwen3_235b_a22b_fp4 precision: fp4 model_dir_name: Qwen3-235B-A22B-FP4 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL.yaml index 5f3cbd8ea60b..accda03e6afd 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP4 + model_name: qwen3_235b_a22b_fp4 precision: fp4 model_dir_name: Qwen3-235B-A22B-FP4 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con4_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con4_ccb-UCX.yaml index 270e6a3b96e8..6ea24bb48100 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con4_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con4_ccb-UCX.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP4 + model_name: qwen3_235b_a22b_fp4 precision: fp4 model_dir_name: Qwen3-235B-A22B-FP4 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL.yaml index 18e0c5bd4da3..338c4d0fbbc0 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP4 + model_name: qwen3_235b_a22b_fp4 precision: fp4 model_dir_name: Qwen3-235B-A22B-FP4 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con8_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con8_ccb-UCX.yaml index e7c3063804d5..1512c2754bd0 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con8_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx1_gen4_tep8_bs32_eplb0_mtp0_con8_ccb-UCX.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP4 + model_name: qwen3_235b_a22b_fp4 precision: fp4 model_dir_name: Qwen3-235B-A22B-FP4 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_con2048_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_con2048_ccb-NIXL.yaml index eb9fd647b7dc..00415ca3564e 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_con2048_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_con2048_ccb-NIXL.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP4 + model_name: qwen3_235b_a22b_fp4 precision: fp4 model_dir_name: Qwen3-235B-A22B-FP4 supported_gpus: @@ -13,7 +13,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: e2e @@ -36,8 +36,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_con2048_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_con2048_ccb-UCX.yaml index e468e51c7eac..6667a05b7532 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_con2048_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb0_mtp1_con2048_ccb-UCX.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP4 + model_name: qwen3_235b_a22b_fp4 precision: fp4 model_dir_name: Qwen3-235B-A22B-FP4 supported_gpus: @@ -13,7 +13,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: e2e @@ -36,8 +36,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con16_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con16_ccb-NIXL.yaml index 49af3a3259d3..b9812cd4eaa5 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con16_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con16_ccb-NIXL.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP8 + model_name: qwen3_235b_a22b_fp8 precision: fp8 model_dir_name: Qwen3-235B-A22B-FP8 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con16_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con16_ccb-UCX.yaml index 9ee8026bbefb..8b2ef1a5cd93 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con16_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con16_ccb-UCX.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP8 + model_name: qwen3_235b_a22b_fp8 precision: fp8 model_dir_name: Qwen3-235B-A22B-FP8 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL.yaml index 321cbf911a0b..56502cbcd9cc 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con1_ccb-NIXL.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP8 + model_name: qwen3_235b_a22b_fp8 precision: fp8 model_dir_name: Qwen3-235B-A22B-FP8 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con1_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con1_ccb-UCX.yaml index 7a7fb603d08c..0cda2dd36331 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con1_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con1_ccb-UCX.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP8 + model_name: qwen3_235b_a22b_fp8 precision: fp8 model_dir_name: Qwen3-235B-A22B-FP8 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL.yaml index ccda3d3e5886..11ba1857debe 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con2_ccb-NIXL.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP8 + model_name: qwen3_235b_a22b_fp8 precision: fp8 model_dir_name: Qwen3-235B-A22B-FP8 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con2_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con2_ccb-UCX.yaml index 86ad52efb5a2..9ebbec0b0f78 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con2_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con2_ccb-UCX.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP8 + model_name: qwen3_235b_a22b_fp8 precision: fp8 model_dir_name: Qwen3-235B-A22B-FP8 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con36_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con36_ccb-NIXL.yaml index d4f58c3cb47c..8988a5c74b4b 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con36_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con36_ccb-NIXL.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP8 + model_name: qwen3_235b_a22b_fp8 precision: fp8 model_dir_name: Qwen3-235B-A22B-FP8 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con36_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con36_ccb-UCX.yaml index f10b8f473470..0e8277f268b8 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con36_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con36_ccb-UCX.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP8 + model_name: qwen3_235b_a22b_fp8 precision: fp8 model_dir_name: Qwen3-235B-A22B-FP8 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL.yaml index 33a3c60b63c7..a5ab0aa447f3 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con4_ccb-NIXL.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP8 + model_name: qwen3_235b_a22b_fp8 precision: fp8 model_dir_name: Qwen3-235B-A22B-FP8 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con4_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con4_ccb-UCX.yaml index e43c2ad1d011..52ec72394d5d 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con4_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con4_ccb-UCX.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP8 + model_name: qwen3_235b_a22b_fp8 precision: fp8 model_dir_name: Qwen3-235B-A22B-FP8 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL.yaml index db8ca23defa9..7657581678a3 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con8_ccb-NIXL.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP8 + model_name: qwen3_235b_a22b_fp8 precision: fp8 model_dir_name: Qwen3-235B-A22B-FP8 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con8_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con8_ccb-UCX.yaml index 4d4958076cc0..bb5c06392f02 100644 --- a/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con8_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/Qwen3-235B-A22B-FP8_1k1k_ctx1_gen1_tep8_bs32_eplb0_mtp0_con8_ccb-UCX.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP8 + model_name: qwen3_235b_a22b_fp8 precision: fp8 model_dir_name: Qwen3-235B-A22B-FP8 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-NIXL.yaml index 74828e6d9a36..54c6b2569c69 100644 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-NIXL.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP4 + model_name: qwen3_235b_a22b_fp4 precision: fp4 model_dir_name: Qwen3-235B-A22B-FP4 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-UCX.yaml index 9630d2c40a40..2850f4dc38ed 100644 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-UCX.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP4 + model_name: qwen3_235b_a22b_fp4 precision: fp4 model_dir_name: Qwen3-235B-A22B-FP4 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-NIXL.yaml index 2e5291955167..5dd51d5299ca 100644 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-NIXL.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP4 + model_name: qwen3_235b_a22b_fp4 precision: fp4 model_dir_name: Qwen3-235B-A22B-FP4 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-UCX.yaml index 248b13a9a8c1..8f4eeb63d0f6 100644 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-UCX.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP4 + model_name: qwen3_235b_a22b_fp4 precision: fp4 model_dir_name: Qwen3-235B-A22B-FP4 supported_gpus: diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml index 02a018e7505a..83ed69ea20dd 100644 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP4 + model_name: qwen3_235b_a22b_fp4 precision: fp4 model_dir_name: Qwen3-235B-A22B-FP4 supported_gpus: @@ -14,7 +14,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: gen_only @@ -37,8 +37,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml index 7ba14aabf6e8..e00b6f103094 100644 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP4 + model_name: qwen3_235b_a22b_fp4 precision: fp4 model_dir_name: Qwen3-235B-A22B-FP4 supported_gpus: @@ -14,7 +14,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: gen_only @@ -37,8 +37,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-NIXL.yaml index b32724a93b2b..943548a8c7dc 100644 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-NIXL.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP4 + model_name: qwen3_235b_a22b_fp4 precision: fp4 model_dir_name: Qwen3-235B-A22B-FP4 supported_gpus: @@ -14,7 +14,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: gen_only @@ -37,8 +37,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-UCX.yaml index 142602f557ca..57ff02f6f46b 100644 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-UCX.yaml @@ -1,5 +1,5 @@ metadata: - model_name: Qwen3-235B-A22B-FP4 + model_name: qwen3_235b_a22b_fp4 precision: fp4 model_dir_name: Qwen3-235B-A22B-FP4 supported_gpus: @@ -14,7 +14,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: gen_only @@ -37,8 +37,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL.yaml index b607270bf5d3..7f926e9119c3 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: @@ -14,7 +14,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: gen_only @@ -37,8 +37,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL_kv-reuse.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL_kv-reuse.yaml index da17daeeb042..ea2d087eda2a 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL_kv-reuse.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL_kv-reuse.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: @@ -14,7 +14,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: gen_only @@ -37,8 +37,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-UCX.yaml index b37f61f92a6f..6ed8299029f9 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-UCX.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: @@ -14,7 +14,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: gen_only @@ -37,8 +37,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-NIXL.yaml index d29762488d0b..c7f6ee259294 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-NIXL.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: @@ -14,7 +14,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: gen_only @@ -37,8 +37,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-UCX.yaml index c5377581b2b7..3de8c372ea56 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-UCX.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: @@ -14,7 +14,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: gen_only @@ -37,8 +37,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_con12288_ccb-DEFAULT.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_con12288_ccb-DEFAULT.yaml index 4d42180625c0..a677c413e7e4 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_con12288_ccb-DEFAULT.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_con12288_ccb-DEFAULT.yaml @@ -1,6 +1,5 @@ -# nvbugs: 5422621 metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: @@ -16,7 +15,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: gen_only @@ -39,8 +38,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_con1024_ccb-DEFAULT.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_con1024_ccb-DEFAULT.yaml index 42ab879197dd..4ca6a4022e53 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_con1024_ccb-DEFAULT.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_con1024_ccb-DEFAULT.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: @@ -14,7 +14,7 @@ slurm: account: job_time: 02:00:00 job_name: disaggr-test - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true hardware: gpus_per_node: 4 @@ -37,8 +37,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-NIXL.yaml index ee4256d4c528..efbc1483c85d 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-NIXL.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: @@ -14,7 +14,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: gen_only @@ -37,8 +37,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-UCX.yaml index 38cd607e96cd..929513aeb6ca 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-UCX.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: @@ -14,7 +14,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: gen_only @@ -37,8 +37,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml index 2fa89f9fff9e..600781067a20 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: @@ -14,7 +14,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: gen_only @@ -37,8 +37,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml index a9ee6c3ed1ae..bf22760695b6 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-r1-fp4 + model_name: deepseek_r1_0528_fp4_v2 precision: fp4 model_dir_name: DeepSeek-R1-0528-FP4-v2 supported_gpus: @@ -14,7 +14,7 @@ slurm: account: job_time: 02:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: gen_only @@ -37,8 +37,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL.yaml index bf9bdc3c113d..5177aa2df085 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-v32-fp4 + model_name: deepseek_v32_fp4 precision: fp4 model_dir_name: DeepSeek-V3.2-FP4-v2 supported_gpus: @@ -15,7 +15,7 @@ slurm: account: job_time: 04:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: gen_only @@ -38,12 +38,13 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: - enable_accuracy_test: false # Set to true to enable accuracy evaluation + enable_accuracy_test: false worker_config: gen: tensor_parallel_size: 32 diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-NIXL.yaml index e65b3e2fa858..71a10fe9d0ab 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-NIXL.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-v32-fp4 + model_name: deepseek_v32_fp4 precision: fp4 model_dir_name: DeepSeek-V3.2-FP4-v2 supported_gpus: @@ -15,7 +15,7 @@ slurm: account: job_time: 04:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: gen_only @@ -38,8 +38,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_con12288_ccb-DEFAULT.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_con12288_ccb-DEFAULT.yaml index 6302846bb6b1..c062c2236775 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_con12288_ccb-DEFAULT.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_con12288_ccb-DEFAULT.yaml @@ -1,6 +1,5 @@ -# nvbugs: 5422621 metadata: - model_name: deepseek-v32-fp4 + model_name: deepseek_v32_fp4 precision: fp4 model_dir_name: DeepSeek-V3.2-FP4-v2 supported_gpus: @@ -16,7 +15,7 @@ slurm: account: job_time: 04:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: gen_only @@ -39,8 +38,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_con1024_ccb-DEFAULT.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_con1024_ccb-DEFAULT.yaml index b37216aee27e..69ea428aeb60 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_con1024_ccb-DEFAULT.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_con1024_ccb-DEFAULT.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-v32-fp4 + model_name: deepseek_v32_fp4 precision: fp4 model_dir_name: DeepSeek-V3.2-FP4-v2 supported_gpus: @@ -15,7 +15,7 @@ slurm: account: job_time: 04:00:00 job_name: disaggr-test - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true hardware: gpus_per_node: 4 @@ -38,8 +38,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-NIXL.yaml index a1f41602e9d1..51e15dc3b35d 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-NIXL.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-v32-fp4 + model_name: deepseek_v32_fp4 precision: fp4 model_dir_name: DeepSeek-V3.2-FP4-v2 supported_gpus: @@ -15,7 +15,7 @@ slurm: account: job_time: 04:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: gen_only @@ -38,8 +38,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml index 6219af7ebc61..8f55177b6c62 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml @@ -1,5 +1,5 @@ metadata: - model_name: deepseek-v32-fp4 + model_name: deepseek_v32_fp4 precision: fp4 model_dir_name: DeepSeek-V3.2-FP4-v2 supported_gpus: @@ -15,7 +15,7 @@ slurm: account: job_time: 04:00:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: gen_only @@ -38,8 +38,9 @@ environment: trtllm_repo: '' build_wheel: false work_dir: - worker_env_var: "TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes" - server_env_var: "TRTLLM_SERVER_DISABLE_GC=1" + worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 + TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes + server_env_var: TRTLLM_SERVER_DISABLE_GC=1 profiling: nsys_on: false accuracy: diff --git a/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_1k1k_ctx3_gen1_dep32_bs1024_eplb384_mtp0_con16384_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_1k1k_ctx3_gen1_dep32_bs1024_eplb384_mtp0_con16384_ccb-NIXL.yaml index aa99610ab548..227415b9b823 100644 --- a/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_1k1k_ctx3_gen1_dep32_bs1024_eplb384_mtp0_con16384_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_1k1k_ctx3_gen1_dep32_bs1024_eplb384_mtp0_con16384_ccb-NIXL.yaml @@ -1,5 +1,5 @@ metadata: - model_name: kimi-k2-thinking-fp4 + model_name: k2_thinking_fp4 precision: fp4 model_dir_name: Kimi-K2-Thinking-NVFP4 supported_gpus: @@ -14,7 +14,7 @@ slurm: account: job_time: 00:45:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: gen_only diff --git a/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_8k1k_ctx8_gen1_dep32_bs256_eplb416_mtp0_con8192_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_8k1k_ctx8_gen1_dep32_bs256_eplb416_mtp0_con8192_ccb-NIXL.yaml index 7c37876ced04..9e220dff94b2 100644 --- a/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_8k1k_ctx8_gen1_dep32_bs256_eplb416_mtp0_con8192_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_8k1k_ctx8_gen1_dep32_bs256_eplb416_mtp0_con8192_ccb-NIXL.yaml @@ -1,5 +1,5 @@ metadata: - model_name: kimi-k2-thinking-fp4 + model_name: k2_thinking_fp4 precision: fp4 model_dir_name: Kimi-K2-Thinking-NVFP4 supported_gpus: @@ -14,7 +14,7 @@ slurm: account: job_time: 00:45:00 job_name: unified-benchmark - extra_args: "--gres=gpu:4" + extra_args: --gres=gpu:4 numa_bind: true benchmark: mode: gen_only From 3d80e0894857a80d7cb6c3db28d951bbc0eab2d6 Mon Sep 17 00:00:00 2001 From: FredricZ-2007 <226039983+fredricz-20070104@users.noreply.github.com> Date: Fri, 6 Mar 2026 10:01:14 +0000 Subject: [PATCH 09/12] update dataset file path Signed-off-by: FredricZ-2007 <226039983+fredricz-20070104@users.noreply.github.com> --- ...k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-NIXL.yaml | 3 +-- ...1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-UCX.yaml | 3 +-- ...1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-NIXL.yaml | 3 +-- ..._1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-UCX.yaml | 3 +-- ...1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml | 3 +-- ..._1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml | 3 +-- ...1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-NIXL.yaml | 3 +-- ...k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-UCX.yaml | 3 +-- ...k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL.yaml | 3 +-- ...gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL_kv-reuse.yaml | 3 +-- ...1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-UCX.yaml | 3 +-- ...1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-NIXL.yaml | 3 +-- ...k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-UCX.yaml | 3 +-- ...ctx2_gen1_dep48_bs16_eplb288_mtp3_con12288_ccb-DEFAULT.yaml | 3 +-- ...ctx2_gen1_dep32_bs128_eplb288_mtp3_con1024_ccb-DEFAULT.yaml | 3 +-- ...k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-NIXL.yaml | 3 +-- ...8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-UCX.yaml | 3 +-- ...8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml | 3 +-- ..._8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml | 3 +-- ...k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL.yaml | 3 +-- ...1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-NIXL.yaml | 3 +-- ...ctx2_gen1_dep48_bs16_eplb288_mtp3_con12288_ccb-DEFAULT.yaml | 3 +-- ...ctx2_gen1_dep32_bs128_eplb288_mtp3_con1024_ccb-DEFAULT.yaml | 3 +-- ...k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-NIXL.yaml | 3 +-- ...8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml | 3 +-- ..._ctx3_gen1_dep32_bs1024_eplb384_mtp0_con16384_ccb-NIXL.yaml | 3 +-- ...1k_ctx8_gen1_dep32_bs256_eplb416_mtp0_con8192_ccb-NIXL.yaml | 3 +-- 27 files changed, 27 insertions(+), 54 deletions(-) diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-NIXL.yaml index 54c6b2569c69..82e2c40e56e9 100644 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-NIXL.yaml @@ -7,7 +7,6 @@ metadata: - GB300 script_file: disaggr_torch.slurm benchmark_type: 1k1k - dataset_file: disagg_datasets/deepseek-r1-1024-1024-100000-ratio-1_for_serve.json slurm: script_file: disaggr_torch.slurm partition: @@ -25,7 +24,7 @@ benchmark: concurrency_list: '1024' input_length: 1024 output_length: 1024 - dataset_file: + dataset_file: datasets/perf-ci/qwen3_235b-1k1k-20480-ratio-1_for_serve.json hardware: gpus_per_node: 4 num_ctx_servers: 1 diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-UCX.yaml index 2850f4dc38ed..d8133fb707d8 100644 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-UCX.yaml @@ -7,7 +7,6 @@ metadata: - GB300 script_file: disaggr_torch.slurm benchmark_type: 1k1k - dataset_file: disagg_datasets/deepseek-r1-1024-1024-100000-ratio-1_for_serve.json slurm: script_file: disaggr_torch.slurm partition: @@ -25,7 +24,7 @@ benchmark: concurrency_list: '1024' input_length: 1024 output_length: 1024 - dataset_file: + dataset_file: datasets/perf-ci/qwen3_235b-1k1k-20480-ratio-1_for_serve.json hardware: gpus_per_node: 4 num_ctx_servers: 1 diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-NIXL.yaml index 5dd51d5299ca..d07eb20c53f1 100644 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-NIXL.yaml @@ -7,7 +7,6 @@ metadata: - GB300 script_file: disaggr_torch.slurm benchmark_type: 1k1k - dataset_file: disagg_datasets/deepseek-r1-1024-1024-100000-ratio-1_for_serve.json slurm: script_file: disaggr_torch.slurm partition: @@ -25,7 +24,7 @@ benchmark: concurrency_list: '512' input_length: 1024 output_length: 1024 - dataset_file: + dataset_file: datasets/perf-ci/qwen3_235b-1k1k-20480-ratio-1_for_serve.json hardware: gpus_per_node: 4 num_ctx_servers: 1 diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-UCX.yaml index 8f4eeb63d0f6..3a1f12957706 100644 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-UCX.yaml @@ -7,7 +7,6 @@ metadata: - GB300 script_file: disaggr_torch.slurm benchmark_type: 1k1k - dataset_file: disagg_datasets/deepseek-r1-1024-1024-100000-ratio-1_for_serve.json slurm: script_file: disaggr_torch.slurm partition: @@ -25,7 +24,7 @@ benchmark: concurrency_list: '512' input_length: 1024 output_length: 1024 - dataset_file: + dataset_file: datasets/perf-ci/qwen3_235b-1k1k-20480-ratio-1_for_serve.json hardware: gpus_per_node: 4 num_ctx_servers: 1 diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml index 83ed69ea20dd..ff376ede72d7 100644 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml @@ -7,7 +7,6 @@ metadata: - GB300 script_file: disaggr_torch.slurm benchmark_type: 1k1k - dataset_file: disagg_datasets/deepseek-r1-1024-1024-100000-ratio-1_for_serve.json slurm: script_file: disaggr_torch.slurm partition: @@ -25,7 +24,7 @@ benchmark: concurrency_list: '512' input_length: 1024 output_length: 1024 - dataset_file: + dataset_file: datasets/perf-ci/qwen3_235b-1k1k-20480-ratio-1_for_serve.json hardware: gpus_per_node: 4 num_ctx_servers: 1 diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml index e00b6f103094..461f766c2cb1 100644 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml @@ -7,7 +7,6 @@ metadata: - GB300 script_file: disaggr_torch.slurm benchmark_type: 1k1k - dataset_file: disagg_datasets/deepseek-r1-1024-1024-100000-ratio-1_for_serve.json slurm: script_file: disaggr_torch.slurm partition: @@ -25,7 +24,7 @@ benchmark: concurrency_list: '512' input_length: 1024 output_length: 1024 - dataset_file: + dataset_file: datasets/perf-ci/qwen3_235b-1k1k-20480-ratio-1_for_serve.json hardware: gpus_per_node: 4 num_ctx_servers: 1 diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-NIXL.yaml index 943548a8c7dc..eda761896272 100644 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-NIXL.yaml @@ -7,7 +7,6 @@ metadata: - GB300 script_file: disaggr_torch.slurm benchmark_type: 1k1k - dataset_file: disagg_datasets/deepseek-r1-1024-1024-100000-ratio-1_for_serve.json slurm: script_file: disaggr_torch.slurm partition: @@ -25,7 +24,7 @@ benchmark: concurrency_list: '2048' input_length: 1024 output_length: 1024 - dataset_file: + dataset_file: datasets/perf-ci/qwen3_235b-1k1k-20480-ratio-1_for_serve.json hardware: gpus_per_node: 4 num_ctx_servers: 2 diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-UCX.yaml index 57ff02f6f46b..d64368504a29 100644 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-UCX.yaml @@ -7,7 +7,6 @@ metadata: - GB300 script_file: disaggr_torch.slurm benchmark_type: 1k1k - dataset_file: disagg_datasets/deepseek-r1-1024-1024-100000-ratio-1_for_serve.json slurm: script_file: disaggr_torch.slurm partition: @@ -25,7 +24,7 @@ benchmark: concurrency_list: '2048' input_length: 1024 output_length: 1024 - dataset_file: + dataset_file: datasets/perf-ci/qwen3_235b-1k1k-20480-ratio-1_for_serve.json hardware: gpus_per_node: 4 num_ctx_servers: 2 diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL.yaml index 7f926e9119c3..ff139ca6b1e4 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL.yaml @@ -7,7 +7,6 @@ metadata: - GB300 script_file: disaggr_torch.slurm benchmark_type: 1k1k - dataset_file: disagg_datasets/deepseek-r1-1024-1024-100000-ratio-1_for_serve.json slurm: script_file: disaggr_torch.slurm partition: @@ -25,7 +24,7 @@ benchmark: concurrency_list: '1024' input_length: 1024 output_length: 1024 - dataset_file: + dataset_file: datasets/perf-ci/deepseek_r1-1k1k-20480-ratio-1_for_serve.json hardware: gpus_per_node: 4 num_ctx_servers: 1 diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL_kv-reuse.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL_kv-reuse.yaml index ea2d087eda2a..60157103167c 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL_kv-reuse.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL_kv-reuse.yaml @@ -7,7 +7,6 @@ metadata: - GB300 script_file: disaggr_torch.slurm benchmark_type: 1k1k - dataset_file: disagg_datasets/deepseek-r1-1024-1024-100000-ratio-1_for_serve.json slurm: script_file: disaggr_torch.slurm partition: @@ -25,7 +24,7 @@ benchmark: concurrency_list: '1024' input_length: 1024 output_length: 1024 - dataset_file: + dataset_file: datasets/perf-ci/deepseek_r1-1k1k-20480-ratio-1_for_serve.json hardware: gpus_per_node: 4 num_ctx_servers: 1 diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-UCX.yaml index 6ed8299029f9..079871ddbc89 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-UCX.yaml @@ -7,7 +7,6 @@ metadata: - GB300 script_file: disaggr_torch.slurm benchmark_type: 1k1k - dataset_file: disagg_datasets/deepseek-r1-1024-1024-100000-ratio-1_for_serve.json slurm: script_file: disaggr_torch.slurm partition: @@ -25,7 +24,7 @@ benchmark: concurrency_list: '1024' input_length: 1024 output_length: 1024 - dataset_file: + dataset_file: datasets/perf-ci/deepseek_r1-1k1k-20480-ratio-1_for_serve.json hardware: gpus_per_node: 4 num_ctx_servers: 1 diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-NIXL.yaml index c7f6ee259294..cacafb92d986 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-NIXL.yaml @@ -7,7 +7,6 @@ metadata: - GB300 script_file: disaggr_torch.slurm benchmark_type: 1k1k - dataset_file: disagg_datasets/deepseek-r1-1024-1024-100000-ratio-1_for_serve.json slurm: script_file: disaggr_torch.slurm partition: @@ -25,7 +24,7 @@ benchmark: concurrency_list: '2048' input_length: 1024 output_length: 1024 - dataset_file: + dataset_file: datasets/perf-ci/deepseek_r1-1k1k-20480-ratio-1_for_serve.json hardware: gpus_per_node: 4 num_ctx_servers: 2 diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-UCX.yaml index 3de8c372ea56..d61f116a6177 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-UCX.yaml @@ -7,7 +7,6 @@ metadata: - GB300 script_file: disaggr_torch.slurm benchmark_type: 1k1k - dataset_file: disagg_datasets/deepseek-r1-1024-1024-100000-ratio-1_for_serve.json slurm: script_file: disaggr_torch.slurm partition: @@ -25,7 +24,7 @@ benchmark: concurrency_list: '2048' input_length: 1024 output_length: 1024 - dataset_file: + dataset_file: datasets/perf-ci/deepseek_r1-1k1k-20480-ratio-1_for_serve.json hardware: gpus_per_node: 4 num_ctx_servers: 2 diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_con12288_ccb-DEFAULT.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_con12288_ccb-DEFAULT.yaml index a677c413e7e4..9a1cb1ac23ae 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_con12288_ccb-DEFAULT.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_con12288_ccb-DEFAULT.yaml @@ -8,7 +8,6 @@ metadata: script_file: disaggr_torch.slurm benchmark_type: 1k1k config_index: 7 - dataset_file: disagg_datasets/deepseek-r1-1024-1024-100000-ratio-1_for_serve.json slurm: script_file: disaggr_torch.slurm partition: @@ -26,7 +25,7 @@ benchmark: concurrency_list: '12288' input_length: 1024 output_length: 1024 - dataset_file: + dataset_file: datasets/perf-ci/deepseek_r1-1k1k-20480-ratio-1_for_serve.json hardware: gpus_per_node: 4 num_ctx_servers: 2 diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_con1024_ccb-DEFAULT.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_con1024_ccb-DEFAULT.yaml index 4ca6a4022e53..9e1996f596b8 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_con1024_ccb-DEFAULT.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_con1024_ccb-DEFAULT.yaml @@ -7,7 +7,6 @@ metadata: - GB300 script_file: disaggr_torch.slurm benchmark_type: 8k1k - dataset_file: disagg_datasets/deepseek-r1-8192-1024-200000-ratio-1_for_serve.json slurm: script_file: disaggr_torch.slurm partition: @@ -29,7 +28,7 @@ benchmark: concurrency_list: '1024' input_length: 8192 output_length: 1024 - dataset_file: + dataset_file: datasets/perf-ci/deepseek_r1-8k1k-20480-ratio-1_for_serve.json environment: container_mount: container_image: diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-NIXL.yaml index efbc1483c85d..31fcc8ebd03b 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-NIXL.yaml @@ -7,7 +7,6 @@ metadata: - GB300 script_file: disaggr_torch.slurm benchmark_type: 8k1k - dataset_file: disagg_datasets/deepseek-r1-8192-1024-200000-ratio-1_for_serve.json slurm: script_file: disaggr_torch.slurm partition: @@ -25,7 +24,7 @@ benchmark: concurrency_list: '1024' input_length: 8192 output_length: 1024 - dataset_file: + dataset_file: datasets/perf-ci/deepseek_r1-8k1k-20480-ratio-1_for_serve.json hardware: gpus_per_node: 4 num_ctx_servers: 6 diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-UCX.yaml index 929513aeb6ca..3433196c31de 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-UCX.yaml @@ -7,7 +7,6 @@ metadata: - GB300 script_file: disaggr_torch.slurm benchmark_type: 8k1k - dataset_file: disagg_datasets/deepseek-r1-8192-1024-200000-ratio-1_for_serve.json slurm: script_file: disaggr_torch.slurm partition: @@ -25,7 +24,7 @@ benchmark: concurrency_list: '1024' input_length: 8192 output_length: 1024 - dataset_file: + dataset_file: datasets/perf-ci/deepseek_r1-8k1k-20480-ratio-1_for_serve.json hardware: gpus_per_node: 4 num_ctx_servers: 6 diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml index 600781067a20..0e6bf4e7d985 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml @@ -7,7 +7,6 @@ metadata: - GB300 script_file: disaggr_torch.slurm benchmark_type: 8k1k - dataset_file: disagg_datasets/deepseek-r1-8192-1024-200000-ratio-1_for_serve.json slurm: script_file: disaggr_torch.slurm partition: @@ -25,7 +24,7 @@ benchmark: concurrency_list: '512' input_length: 8192 output_length: 1024 - dataset_file: + dataset_file: datasets/perf-ci/deepseek_r1-8k1k-20480-ratio-1_for_serve.json hardware: gpus_per_node: 4 num_ctx_servers: 8 diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml index bf22760695b6..b7743c9bd812 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-r1-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml @@ -7,7 +7,6 @@ metadata: - GB300 script_file: disaggr_torch.slurm benchmark_type: 8k1k - dataset_file: disagg_datasets/deepseek-r1-8192-1024-200000-ratio-1_for_serve.json slurm: script_file: disaggr_torch.slurm partition: @@ -25,7 +24,7 @@ benchmark: concurrency_list: '512' input_length: 8192 output_length: 1024 - dataset_file: + dataset_file: datasets/perf-ci/deepseek_r1-8k1k-20480-ratio-1_for_serve.json hardware: gpus_per_node: 4 num_ctx_servers: 8 diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL.yaml index 5177aa2df085..d194178ac64d 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL.yaml @@ -8,7 +8,6 @@ metadata: script_file: disaggr_torch.slurm benchmark_type: 1k1k config_index: 1 - dataset_file: disagg_datasets/deepseek-v32-1024-1024-200000-ratio-1_for_serve.json slurm: script_file: disaggr_torch.slurm partition: @@ -26,7 +25,7 @@ benchmark: concurrency_list: '1024' input_length: 1024 output_length: 1024 - dataset_file: + dataset_file: datasets/perf-ci/deepseek_v32-1k1k-20480-ratio-1_for_serve.json hardware: gpus_per_node: 4 num_ctx_servers: 1 diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-NIXL.yaml index 71a10fe9d0ab..8e4780400a49 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp3_con2048_ccb-NIXL.yaml @@ -8,7 +8,6 @@ metadata: script_file: disaggr_torch.slurm benchmark_type: 1k1k config_index: 0 - dataset_file: disagg_datasets/deepseek-v32-1024-1024-200000-ratio-1_for_serve.json slurm: script_file: disaggr_torch.slurm partition: @@ -26,7 +25,7 @@ benchmark: concurrency_list: '2048' input_length: 1024 output_length: 1024 - dataset_file: + dataset_file: datasets/perf-ci/deepseek_v32-1k1k-20480-ratio-1_for_serve.json hardware: gpus_per_node: 4 num_ctx_servers: 2 diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_con12288_ccb-DEFAULT.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_con12288_ccb-DEFAULT.yaml index c062c2236775..691ffadfd327 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_con12288_ccb-DEFAULT.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_1k1k_ctx2_gen1_dep48_bs16_eplb288_mtp3_con12288_ccb-DEFAULT.yaml @@ -8,7 +8,6 @@ metadata: script_file: disaggr_torch.slurm benchmark_type: 1k1k config_index: 7 - dataset_file: disagg_datasets/deepseek-v32-1024-1024-200000-ratio-1_for_serve.json slurm: script_file: disaggr_torch.slurm partition: @@ -26,7 +25,7 @@ benchmark: concurrency_list: '12288' input_length: 1024 output_length: 1024 - dataset_file: + dataset_file: datasets/perf-ci/deepseek_v32-1k1k-20480-ratio-1_for_serve.json hardware: gpus_per_node: 4 num_ctx_servers: 2 diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_con1024_ccb-DEFAULT.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_con1024_ccb-DEFAULT.yaml index 69ea428aeb60..d2df7e4574f1 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_con1024_ccb-DEFAULT.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx2_gen1_dep32_bs128_eplb288_mtp3_con1024_ccb-DEFAULT.yaml @@ -8,7 +8,6 @@ metadata: script_file: disaggr_torch.slurm benchmark_type: 8k1k config_index: 14 - dataset_file: disagg_datasets/deepseek-v32-8192-1024-200000-ratio-1_for_serve.json slurm: script_file: disaggr_torch.slurm partition: @@ -30,7 +29,7 @@ benchmark: concurrency_list: '1024' input_length: 8192 output_length: 1024 - dataset_file: + dataset_file: datasets/perf-ci/deepseek_v32-8k1k-20480-ratio-1_for_serve.json environment: container_mount: container_image: diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-NIXL.yaml index 51e15dc3b35d..a7723ba302fd 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx6_gen1_dep16_bs64_eplb288_mtp0_con1024_ccb-NIXL.yaml @@ -8,7 +8,6 @@ metadata: script_file: disaggr_torch.slurm benchmark_type: 8k1k config_index: 5 - dataset_file: disagg_datasets/deepseek-v32-8192-1024-200000-ratio-1_for_serve.json slurm: script_file: disaggr_torch.slurm partition: @@ -26,7 +25,7 @@ benchmark: concurrency_list: '1024' input_length: 8192 output_length: 1024 - dataset_file: + dataset_file: datasets/perf-ci/deepseek_v32-8k1k-20480-ratio-1_for_serve.json hardware: gpus_per_node: 4 num_ctx_servers: 6 diff --git a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml index 8f55177b6c62..a2df3a17555d 100644 --- a/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_deepseek-v32-fp4_8k1k_ctx8_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml @@ -8,7 +8,6 @@ metadata: script_file: disaggr_torch.slurm benchmark_type: 8k1k config_index: 4 - dataset_file: disagg_datasets/deepseek-v32-8192-1024-200000-ratio-1_for_serve.json slurm: script_file: disaggr_torch.slurm partition: @@ -26,7 +25,7 @@ benchmark: concurrency_list: '512' input_length: 8192 output_length: 1024 - dataset_file: + dataset_file: datasets/perf-ci/deepseek_v32-8k1k-20480-ratio-1_for_serve.json hardware: gpus_per_node: 4 num_ctx_servers: 8 diff --git a/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_1k1k_ctx3_gen1_dep32_bs1024_eplb384_mtp0_con16384_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_1k1k_ctx3_gen1_dep32_bs1024_eplb384_mtp0_con16384_ccb-NIXL.yaml index 227415b9b823..1d2e6f73d5ec 100644 --- a/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_1k1k_ctx3_gen1_dep32_bs1024_eplb384_mtp0_con16384_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_1k1k_ctx3_gen1_dep32_bs1024_eplb384_mtp0_con16384_ccb-NIXL.yaml @@ -7,7 +7,6 @@ metadata: - GB300 script_file: disaggr_torch.slurm benchmark_type: 1k1k - dataset_file: disagg_datasets/kimi-k2-1024-1024-20000-ratio-1_for_serve.json slurm: script_file: disaggr_torch.slurm partition: @@ -25,7 +24,7 @@ benchmark: concurrency_list: '16384' input_length: 1024 output_length: 1024 - dataset_file: + dataset_file: datasets/perf-ci/k2_thinking-1k1k-20480-ratio-1_for_serve.json hardware: gpus_per_node: 4 num_ctx_servers: 3 diff --git a/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_8k1k_ctx8_gen1_dep32_bs256_eplb416_mtp0_con8192_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_8k1k_ctx8_gen1_dep32_bs256_eplb416_mtp0_con8192_ccb-NIXL.yaml index 9e220dff94b2..d421bcd7ebe2 100644 --- a/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_8k1k_ctx8_gen1_dep32_bs256_eplb416_mtp0_con8192_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/wideep_kimi-k2-thinking-fp4_8k1k_ctx8_gen1_dep32_bs256_eplb416_mtp0_con8192_ccb-NIXL.yaml @@ -7,7 +7,6 @@ metadata: - GB300 script_file: disaggr_torch.slurm benchmark_type: 8k1k - dataset_file: disagg_datasets/kimi-k2-8192-1024-20000-ratio-1_for_serve.json slurm: script_file: disaggr_torch.slurm partition: @@ -25,7 +24,7 @@ benchmark: concurrency_list: '8192' input_length: 8192 output_length: 1024 - dataset_file: + dataset_file: datasets/perf-ci/k2_thinking-8k1k-20480-ratio-1_for_serve.json hardware: gpus_per_node: 4 num_ctx_servers: 8 From 6a75ec1ba08efab5e1f16beae7e5eefb6214c93f Mon Sep 17 00:00:00 2001 From: FredricZ-2007 <226039983+fredricz-20070104@users.noreply.github.com> Date: Mon, 9 Mar 2026 03:02:53 +0000 Subject: [PATCH 10/12] remove reduntunt qwen moe cases Signed-off-by: FredricZ-2007 <226039983+fredricz-20070104@users.noreply.github.com> --- .../test_lists/qa/llm_perf_multinode.txt | 16 +-- ...16_bs64_eplb288_mtp3_con1024_ccb-NIXL.yaml | 113 ------------------ ...p16_bs64_eplb288_mtp3_con1024_ccb-UCX.yaml | 113 ------------------ ...p16_bs64_eplb288_mtp3_con512_ccb-NIXL.yaml | 113 ------------------ ...ep16_bs64_eplb288_mtp3_con512_ccb-UCX.yaml | 113 ------------------ ...p32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml | 113 ------------------ ...ep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml | 113 ------------------ ...6_bs128_eplb288_mtp1_con2048_ccb-NIXL.yaml | 113 ------------------ ...16_bs128_eplb288_mtp1_con2048_ccb-UCX.yaml | 113 ------------------ 9 files changed, 8 insertions(+), 912 deletions(-) delete mode 100644 tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-NIXL.yaml delete mode 100644 tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-UCX.yaml delete mode 100644 tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-NIXL.yaml delete mode 100644 tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-UCX.yaml delete mode 100644 tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml delete mode 100644 tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml delete mode 100644 tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-NIXL.yaml delete mode 100644 tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-UCX.yaml diff --git a/tests/integration/test_lists/qa/llm_perf_multinode.txt b/tests/integration/test_lists/qa/llm_perf_multinode.txt index 5499073e29d3..b333f36ee34b 100644 --- a/tests/integration/test_lists/qa/llm_perf_multinode.txt +++ b/tests/integration/test_lists/qa/llm_perf_multinode.txt @@ -131,14 +131,14 @@ perf/test_perf_sanity.py::test_e2e[disagg-e2e-deepseek-r1-fp4_128k8k_ctx8_pp4_ge # wideep multi-node # GB200 + GB300 supported cases -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-UCX] -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-UCX] -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX] -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-NIXL] -perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-UCX] +# perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-NIXL] +# perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-NIXL] +# perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-UCX] +# perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-UCX] +# perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL] +# perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX] +# perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-NIXL] +# perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-UCX] perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL] perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-NIXL_kv-reuse] perf/test_perf_sanity.py::test_e2e[disagg-gen_only-wideep_deepseek-r1-fp4_1k1k_ctx1_gen1_dep32_bs32_eplb288_mtp0_con1024_ccb-UCX] diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-NIXL.yaml deleted file mode 100644 index 82e2c40e56e9..000000000000 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-NIXL.yaml +++ /dev/null @@ -1,113 +0,0 @@ -metadata: - model_name: qwen3_235b_a22b_fp4 - precision: fp4 - model_dir_name: Qwen3-235B-A22B-FP4 - supported_gpus: - - GB200 - - GB300 - script_file: disaggr_torch.slurm - benchmark_type: 1k1k -slurm: - script_file: disaggr_torch.slurm - partition: - account: - job_time: 02:00:00 - job_name: unified-benchmark - extra_args: --gres=gpu:4 - numa_bind: true -benchmark: - mode: gen_only - use_nv_sa_benchmark: false - multi_round: 1 - benchmark_ratio: 0.8 - streaming: true - concurrency_list: '1024' - input_length: 1024 - output_length: 1024 - dataset_file: datasets/perf-ci/qwen3_235b-1k1k-20480-ratio-1_for_serve.json -hardware: - gpus_per_node: 4 - num_ctx_servers: 1 - num_gen_servers: 1 -environment: - container_mount: - container_image: - model_path: - trtllm_repo: '' - build_wheel: false - work_dir: - worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 - TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes - server_env_var: TRTLLM_SERVER_DISABLE_GC=1 -profiling: - nsys_on: false -accuracy: - enable_accuracy_test: false -worker_config: - gen: - enable_layerwise_nvtx_marker: true - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - pipeline_parallel_size: 1 - max_batch_size: 64 - max_num_tokens: 256 - max_seq_len: 2251 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 32 - - 64 - - 128 - - 256 - - 512 - - 768 - - 1024 - - 2048 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.7 - dtype: fp8 - moe_config: - backend: WIDEEP - use_low_precision_moe_combine: true - load_balancer: - num_slots: 288 - layer_updates_per_iter: 1 - cache_transceiver_config: - max_tokens_in_buffer: 4608 - backend: NIXL - stream_interval: 20 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - ctx: - enable_layerwise_nvtx_marker: true - max_batch_size: 4 - max_num_tokens: 4608 - max_seq_len: 2251 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: null - disable_overlap_scheduler: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 4608 - backend: NIXL - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-UCX.yaml deleted file mode 100644 index d8133fb707d8..000000000000 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con1024_ccb-UCX.yaml +++ /dev/null @@ -1,113 +0,0 @@ -metadata: - model_name: qwen3_235b_a22b_fp4 - precision: fp4 - model_dir_name: Qwen3-235B-A22B-FP4 - supported_gpus: - - GB200 - - GB300 - script_file: disaggr_torch.slurm - benchmark_type: 1k1k -slurm: - script_file: disaggr_torch.slurm - partition: - account: - job_time: 02:00:00 - job_name: unified-benchmark - extra_args: --gres=gpu:4 - numa_bind: true -benchmark: - mode: gen_only - use_nv_sa_benchmark: false - multi_round: 1 - benchmark_ratio: 0.8 - streaming: true - concurrency_list: '1024' - input_length: 1024 - output_length: 1024 - dataset_file: datasets/perf-ci/qwen3_235b-1k1k-20480-ratio-1_for_serve.json -hardware: - gpus_per_node: 4 - num_ctx_servers: 1 - num_gen_servers: 1 -environment: - container_mount: - container_image: - model_path: - trtllm_repo: '' - build_wheel: false - work_dir: - worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 - TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes - server_env_var: TRTLLM_SERVER_DISABLE_GC=1 -profiling: - nsys_on: false -accuracy: - enable_accuracy_test: false -worker_config: - gen: - enable_layerwise_nvtx_marker: true - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - pipeline_parallel_size: 1 - max_batch_size: 64 - max_num_tokens: 256 - max_seq_len: 2251 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 32 - - 64 - - 128 - - 256 - - 512 - - 768 - - 1024 - - 2048 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.7 - dtype: fp8 - moe_config: - backend: WIDEEP - use_low_precision_moe_combine: true - load_balancer: - num_slots: 288 - layer_updates_per_iter: 1 - cache_transceiver_config: - max_tokens_in_buffer: 4608 - backend: UCX - stream_interval: 20 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - ctx: - enable_layerwise_nvtx_marker: true - max_batch_size: 4 - max_num_tokens: 4608 - max_seq_len: 2251 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: null - disable_overlap_scheduler: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 4608 - backend: UCX - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-NIXL.yaml deleted file mode 100644 index d07eb20c53f1..000000000000 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-NIXL.yaml +++ /dev/null @@ -1,113 +0,0 @@ -metadata: - model_name: qwen3_235b_a22b_fp4 - precision: fp4 - model_dir_name: Qwen3-235B-A22B-FP4 - supported_gpus: - - GB200 - - GB300 - script_file: disaggr_torch.slurm - benchmark_type: 1k1k -slurm: - script_file: disaggr_torch.slurm - partition: - account: - job_time: 02:00:00 - job_name: unified-benchmark - extra_args: --gres=gpu:4 - numa_bind: true -benchmark: - mode: gen_only - use_nv_sa_benchmark: false - multi_round: 1 - benchmark_ratio: 0.8 - streaming: true - concurrency_list: '512' - input_length: 1024 - output_length: 1024 - dataset_file: datasets/perf-ci/qwen3_235b-1k1k-20480-ratio-1_for_serve.json -hardware: - gpus_per_node: 4 - num_ctx_servers: 1 - num_gen_servers: 1 -environment: - container_mount: - container_image: - model_path: - trtllm_repo: '' - build_wheel: false - work_dir: - worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 - TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes - server_env_var: TRTLLM_SERVER_DISABLE_GC=1 -profiling: - nsys_on: false -accuracy: - enable_accuracy_test: false -worker_config: - gen: - enable_layerwise_nvtx_marker: true - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - pipeline_parallel_size: 1 - max_batch_size: 64 - max_num_tokens: 256 - max_seq_len: 2251 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 32 - - 64 - - 128 - - 256 - - 512 - - 768 - - 1024 - - 2048 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.7 - dtype: fp8 - moe_config: - backend: WIDEEP - use_low_precision_moe_combine: true - load_balancer: - num_slots: 288 - layer_updates_per_iter: 1 - cache_transceiver_config: - max_tokens_in_buffer: 4608 - backend: NIXL - stream_interval: 20 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - ctx: - enable_layerwise_nvtx_marker: true - max_batch_size: 4 - max_num_tokens: 4608 - max_seq_len: 2251 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: null - disable_overlap_scheduler: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 4608 - backend: NIXL - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-UCX.yaml deleted file mode 100644 index 3a1f12957706..000000000000 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep16_bs64_eplb288_mtp3_con512_ccb-UCX.yaml +++ /dev/null @@ -1,113 +0,0 @@ -metadata: - model_name: qwen3_235b_a22b_fp4 - precision: fp4 - model_dir_name: Qwen3-235B-A22B-FP4 - supported_gpus: - - GB200 - - GB300 - script_file: disaggr_torch.slurm - benchmark_type: 1k1k -slurm: - script_file: disaggr_torch.slurm - partition: - account: - job_time: 02:00:00 - job_name: unified-benchmark - extra_args: --gres=gpu:4 - numa_bind: true -benchmark: - mode: gen_only - use_nv_sa_benchmark: false - multi_round: 1 - benchmark_ratio: 0.8 - streaming: true - concurrency_list: '512' - input_length: 1024 - output_length: 1024 - dataset_file: datasets/perf-ci/qwen3_235b-1k1k-20480-ratio-1_for_serve.json -hardware: - gpus_per_node: 4 - num_ctx_servers: 1 - num_gen_servers: 1 -environment: - container_mount: - container_image: - model_path: - trtllm_repo: '' - build_wheel: false - work_dir: - worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 - TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes - server_env_var: TRTLLM_SERVER_DISABLE_GC=1 -profiling: - nsys_on: false -accuracy: - enable_accuracy_test: false -worker_config: - gen: - enable_layerwise_nvtx_marker: true - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - pipeline_parallel_size: 1 - max_batch_size: 64 - max_num_tokens: 256 - max_seq_len: 2251 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 32 - - 64 - - 128 - - 256 - - 512 - - 768 - - 1024 - - 2048 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.7 - dtype: fp8 - moe_config: - backend: WIDEEP - use_low_precision_moe_combine: true - load_balancer: - num_slots: 288 - layer_updates_per_iter: 1 - cache_transceiver_config: - max_tokens_in_buffer: 4608 - backend: UCX - stream_interval: 20 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - ctx: - enable_layerwise_nvtx_marker: true - max_batch_size: 4 - max_num_tokens: 4608 - max_seq_len: 2251 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: null - disable_overlap_scheduler: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 4608 - backend: UCX - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml deleted file mode 100644 index ff376ede72d7..000000000000 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-NIXL.yaml +++ /dev/null @@ -1,113 +0,0 @@ -metadata: - model_name: qwen3_235b_a22b_fp4 - precision: fp4 - model_dir_name: Qwen3-235B-A22B-FP4 - supported_gpus: - - GB200 - - GB300 - script_file: disaggr_torch.slurm - benchmark_type: 1k1k -slurm: - script_file: disaggr_torch.slurm - partition: - account: - job_time: 02:00:00 - job_name: unified-benchmark - extra_args: --gres=gpu:4 - numa_bind: true -benchmark: - mode: gen_only - use_nv_sa_benchmark: false - multi_round: 1 - benchmark_ratio: 0.8 - streaming: true - concurrency_list: '512' - input_length: 1024 - output_length: 1024 - dataset_file: datasets/perf-ci/qwen3_235b-1k1k-20480-ratio-1_for_serve.json -hardware: - gpus_per_node: 4 - num_ctx_servers: 1 - num_gen_servers: 1 -environment: - container_mount: - container_image: - model_path: - trtllm_repo: '' - build_wheel: false - work_dir: - worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 - TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes - server_env_var: TRTLLM_SERVER_DISABLE_GC=1 -profiling: - nsys_on: false -accuracy: - enable_accuracy_test: false -worker_config: - gen: - enable_layerwise_nvtx_marker: true - tensor_parallel_size: 32 - moe_expert_parallel_size: 32 - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - pipeline_parallel_size: 1 - max_batch_size: 16 - max_num_tokens: 64 - max_seq_len: 2251 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 32 - - 64 - - 128 - - 256 - - 512 - - 768 - - 1024 - - 2048 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - dtype: fp8 - moe_config: - backend: WIDEEP - use_low_precision_moe_combine: true - load_balancer: - num_slots: 288 - layer_updates_per_iter: 1 - cache_transceiver_config: - max_tokens_in_buffer: 4608 - backend: NIXL - stream_interval: 20 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - ctx: - enable_layerwise_nvtx_marker: true - max_batch_size: 4 - max_num_tokens: 4608 - max_seq_len: 2251 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: null - disable_overlap_scheduler: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 4608 - backend: NIXL - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml deleted file mode 100644 index 461f766c2cb1..000000000000 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx1_gen1_dep32_bs16_eplb288_mtp3_con512_ccb-UCX.yaml +++ /dev/null @@ -1,113 +0,0 @@ -metadata: - model_name: qwen3_235b_a22b_fp4 - precision: fp4 - model_dir_name: Qwen3-235B-A22B-FP4 - supported_gpus: - - GB200 - - GB300 - script_file: disaggr_torch.slurm - benchmark_type: 1k1k -slurm: - script_file: disaggr_torch.slurm - partition: - account: - job_time: 02:00:00 - job_name: unified-benchmark - extra_args: --gres=gpu:4 - numa_bind: true -benchmark: - mode: gen_only - use_nv_sa_benchmark: false - multi_round: 1 - benchmark_ratio: 0.8 - streaming: true - concurrency_list: '512' - input_length: 1024 - output_length: 1024 - dataset_file: datasets/perf-ci/qwen3_235b-1k1k-20480-ratio-1_for_serve.json -hardware: - gpus_per_node: 4 - num_ctx_servers: 1 - num_gen_servers: 1 -environment: - container_mount: - container_image: - model_path: - trtllm_repo: '' - build_wheel: false - work_dir: - worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 - TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes - server_env_var: TRTLLM_SERVER_DISABLE_GC=1 -profiling: - nsys_on: false -accuracy: - enable_accuracy_test: false -worker_config: - gen: - enable_layerwise_nvtx_marker: true - tensor_parallel_size: 32 - moe_expert_parallel_size: 32 - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - pipeline_parallel_size: 1 - max_batch_size: 16 - max_num_tokens: 64 - max_seq_len: 2251 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 32 - - 64 - - 128 - - 256 - - 512 - - 768 - - 1024 - - 2048 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - dtype: fp8 - moe_config: - backend: WIDEEP - use_low_precision_moe_combine: true - load_balancer: - num_slots: 288 - layer_updates_per_iter: 1 - cache_transceiver_config: - max_tokens_in_buffer: 4608 - backend: UCX - stream_interval: 20 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - ctx: - enable_layerwise_nvtx_marker: true - max_batch_size: 4 - max_num_tokens: 4608 - max_seq_len: 2251 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: null - disable_overlap_scheduler: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 4608 - backend: UCX - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-NIXL.yaml deleted file mode 100644 index eda761896272..000000000000 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-NIXL.yaml +++ /dev/null @@ -1,113 +0,0 @@ -metadata: - model_name: qwen3_235b_a22b_fp4 - precision: fp4 - model_dir_name: Qwen3-235B-A22B-FP4 - supported_gpus: - - GB200 - - GB300 - script_file: disaggr_torch.slurm - benchmark_type: 1k1k -slurm: - script_file: disaggr_torch.slurm - partition: - account: - job_time: 02:00:00 - job_name: unified-benchmark - extra_args: --gres=gpu:4 - numa_bind: true -benchmark: - mode: gen_only - use_nv_sa_benchmark: false - multi_round: 1 - benchmark_ratio: 0.8 - streaming: true - concurrency_list: '2048' - input_length: 1024 - output_length: 1024 - dataset_file: datasets/perf-ci/qwen3_235b-1k1k-20480-ratio-1_for_serve.json -hardware: - gpus_per_node: 4 - num_ctx_servers: 2 - num_gen_servers: 1 -environment: - container_mount: - container_image: - model_path: - trtllm_repo: '' - build_wheel: false - work_dir: - worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 - TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes - server_env_var: TRTLLM_SERVER_DISABLE_GC=1 -profiling: - nsys_on: false -accuracy: - enable_accuracy_test: false -worker_config: - gen: - enable_layerwise_nvtx_marker: true - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - pipeline_parallel_size: 1 - max_batch_size: 128 - max_num_tokens: 256 - max_seq_len: 2251 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 32 - - 64 - - 128 - - 256 - - 512 - - 768 - - 1024 - - 2048 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.7 - dtype: fp8 - moe_config: - backend: WIDEEP - use_low_precision_moe_combine: true - load_balancer: - num_slots: 288 - layer_updates_per_iter: 1 - cache_transceiver_config: - max_tokens_in_buffer: 4608 - backend: NIXL - stream_interval: 20 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 1 - ctx: - enable_layerwise_nvtx_marker: true - max_batch_size: 4 - max_num_tokens: 4608 - max_seq_len: 2251 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: null - disable_overlap_scheduler: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 4608 - backend: NIXL - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 1 diff --git a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-UCX.yaml b/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-UCX.yaml deleted file mode 100644 index d64368504a29..000000000000 --- a/tests/scripts/perf/disaggregated/wideep_Qwen3-235B-A22B-FP4_1k1k_ctx2_gen1_dep16_bs128_eplb288_mtp1_con2048_ccb-UCX.yaml +++ /dev/null @@ -1,113 +0,0 @@ -metadata: - model_name: qwen3_235b_a22b_fp4 - precision: fp4 - model_dir_name: Qwen3-235B-A22B-FP4 - supported_gpus: - - GB200 - - GB300 - script_file: disaggr_torch.slurm - benchmark_type: 1k1k -slurm: - script_file: disaggr_torch.slurm - partition: - account: - job_time: 02:00:00 - job_name: unified-benchmark - extra_args: --gres=gpu:4 - numa_bind: true -benchmark: - mode: gen_only - use_nv_sa_benchmark: false - multi_round: 1 - benchmark_ratio: 0.8 - streaming: true - concurrency_list: '2048' - input_length: 1024 - output_length: 1024 - dataset_file: datasets/perf-ci/qwen3_235b-1k1k-20480-ratio-1_for_serve.json -hardware: - gpus_per_node: 4 - num_ctx_servers: 2 - num_gen_servers: 1 -environment: - container_mount: - container_image: - model_path: - trtllm_repo: '' - build_wheel: false - work_dir: - worker_env_var: TLLM_LOG_LEVEL=INFO TRTLLM_SERVER_DISABLE_GC=1 TRTLLM_WORKER_DISABLE_GC=1 - TRTLLM_ENABLE_PDL=1 ENROOT_ALLOW_DEV=yes - server_env_var: TRTLLM_SERVER_DISABLE_GC=1 -profiling: - nsys_on: false -accuracy: - enable_accuracy_test: false -worker_config: - gen: - enable_layerwise_nvtx_marker: true - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - pipeline_parallel_size: 1 - max_batch_size: 128 - max_num_tokens: 256 - max_seq_len: 2251 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 32 - - 64 - - 128 - - 256 - - 512 - - 768 - - 1024 - - 2048 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.7 - dtype: fp8 - moe_config: - backend: WIDEEP - use_low_precision_moe_combine: true - load_balancer: - num_slots: 288 - layer_updates_per_iter: 1 - cache_transceiver_config: - max_tokens_in_buffer: 4608 - backend: UCX - stream_interval: 20 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 1 - ctx: - enable_layerwise_nvtx_marker: true - max_batch_size: 4 - max_num_tokens: 4608 - max_seq_len: 2251 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: null - disable_overlap_scheduler: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 4608 - backend: UCX - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 1 From 7eeae5e16c53d0f013417de30aa83a030983d82b Mon Sep 17 00:00:00 2001 From: FredricZ-2007 <226039983+fredricz-20070104@users.noreply.github.com> Date: Mon, 9 Mar 2026 05:25:43 +0000 Subject: [PATCH 11/12] use main's waive file Signed-off-by: FredricZ-2007 <226039983+fredricz-20070104@users.noreply.github.com> --- tests/integration/test_lists/waives.txt | 4 ---- 1 file changed, 4 deletions(-) diff --git a/tests/integration/test_lists/waives.txt b/tests/integration/test_lists/waives.txt index 9eddb0639e08..1abcb7a0b3a2 100644 --- a/tests/integration/test_lists/waives.txt +++ b/tests/integration/test_lists/waives.txt @@ -279,11 +279,7 @@ accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_w4_1gpu[v1_kv_cache-True-True accuracy/test_llm_api_pytorch.py::TestQwen3_30B_A3B::test_fp8[latency-torch_compile=False] SKIP (https://nvbugs/5863806) accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_w4_4gpus[v1_kv_cache-dp4-trtllm-auto] SKIP (https://nvbugs/5596343) test_e2e.py::test_trtllm_multimodal_benchmark_serving SKIP (https://nvbugs/5864769) -<<<<<<< HEAD -unittest/_torch/auto_deploy/unit/multigpu/transformations/library/test_bmm_sharding.py::test_sharding[1-1] SKIP (https://nvbugs/5875203) -======= unittest/auto_deploy/multigpu/transformations/library/test_bmm_sharding.py::test_sharding[1-1] SKIP (https://nvbugs/5875203) ->>>>>>> upstream/main accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_fp8_block_scales[mtp=vanilla-fp8kv=False-attention_dp=False-cuda_graph=False-overlap_scheduler=False-torch_compile=False] SKIP (https://nvbugs/5879577) accuracy/test_llm_api_pytorch.py::TestMiniMaxM2::test_4gpus[attention_dp=False-cuda_graph=True-overlap_scheduler=True-tp_size=4-ep_size=4] SKIP (https://nvbugs/5879588) accuracy/test_llm_api_pytorch.py::TestNemotronV3Super::test_auto_dtype_4gpus[4-1-False-False-False] SKIP (https://nvbugs/5879625) From 0a553e8f25193413cefa3b7c5f56e4faae46458c Mon Sep 17 00:00:00 2001 From: FredricZ-2007 <226039983+fredricz-20070104@users.noreply.github.com> Date: Mon, 9 Mar 2026 06:57:05 +0000 Subject: [PATCH 12/12] fx pre-commit error Signed-off-by: FredricZ-2007 <226039983+fredricz-20070104@users.noreply.github.com> --- tests/integration/defs/perf/test_perf_sanity.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/integration/defs/perf/test_perf_sanity.py b/tests/integration/defs/perf/test_perf_sanity.py index 66b7e9a41843..88d46bf85964 100644 --- a/tests/integration/defs/perf/test_perf_sanity.py +++ b/tests/integration/defs/perf/test_perf_sanity.py @@ -55,7 +55,7 @@ "gpt_oss_120b_fp4": "gpt_oss/gpt-oss-120b", "k2_thinking_fp4": "Kimi-K2-Thinking-NVFP4", "qwen3_235b_a22b_fp4": "Qwen3/saved_models_Qwen3-235B-A22B_nvfp4_hf", # Qwen3-235B-A22B-FP4 - "qwen3_235b_a22b_fp8": "Qwen3/saved_models_Qwen3-235B-A22B_fp8_hf", # Qwen3-235B-A22B-FP8 + "qwen3_235b_a22b_fp8": "Qwen3/saved_models_Qwen3-235B-A22B_fp8_hf", # Qwen3-235B-A22B-FP8 } SUPPORTED_GPU_MAPPING = {