diff --git a/tests/integration/defs/.test_durations b/tests/integration/defs/.test_durations index 30d5b9d69b81..d16adb97bb1c 100644 --- a/tests/integration/defs/.test_durations +++ b/tests/integration/defs/.test_durations @@ -74,7 +74,6 @@ "accuracy/test_llm_api_autodeploy.py::TestModelRegistryAccuracy::test_autodeploy_from_registry[deepseek-ai_DeepSeek-R1-0528-True]": 807.0042, "accuracy/test_llm_api_autodeploy.py::TestModelRegistryAccuracy::test_autodeploy_from_registry[google_gemma-3-1b-it-False]": 54.751599999999996, "accuracy/test_llm_api_autodeploy.py::TestModelRegistryAccuracy::test_autodeploy_from_registry[meta-llama_Llama-3.1-8B-Instruct-False]": 43.825, - "accuracy/test_llm_api_autodeploy.py::TestModelRegistryAccuracy::test_autodeploy_from_registry[meta-llama_Llama-3.3-70B-Instruct-False]": 141.4636, "accuracy/test_llm_api_autodeploy.py::TestModelRegistryAccuracy::test_autodeploy_from_registry[mistralai_Codestral-22B-v0.1-False]": 85.2512, "accuracy/test_llm_api_autodeploy.py::TestModelRegistryAccuracy::test_autodeploy_from_registry[mistralai_Ministral-8B-Instruct-2410-False]": 58.8104, "accuracy/test_llm_api_autodeploy.py::TestModelRegistryAccuracy::test_autodeploy_from_registry[nvidia_DeepSeek-R1-0528-NVFP4-v2-True]": 1474.047, @@ -461,11 +460,6 @@ "accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_pard[overlap_scheduler=False]": 774.0882071823204, "accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_pard[overlap_scheduler=True]": 699.8562759562841, "accuracy/test_llm_api_pytorch.py::TestLlama3_1_8B_Instruct_RocketKV::test_auto_dtype": 922.8317, - "accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp4_tp2pp2[torch_compile=True-enable_gemm_allreduce_fusion=False]": 535.9970232558139, - "accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp8_tp4[torch_compile=False]": 607.9050751879699, - "accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp8_tp4[torch_compile=True]": 807.2856390977444, - "accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_nvfp4_tp4[torch_compile=False]": 633.9083157894737, - "accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_nvfp4_tp4[torch_compile=True]": 815.445175572519, "accuracy/test_llm_api_pytorch.py::TestMiniMaxM2::test_4gpus[attention_dp=False-cuda_graph=True-overlap_scheduler=True-tp_size=4-ep_size=4]": 292.3591666666667, "accuracy/test_llm_api_pytorch.py::TestMiniMaxM3::test_nvfp4[use_msa=True]": 863.916775510204, "accuracy/test_llm_api_pytorch.py::TestMistralLarge3_675B::test_nvfp4_4gpus[latency_moe_trtllm]": 476.5565, @@ -917,7 +911,6 @@ "perf/test_perf_sanity.py::test_e2e[aggr_upload-gpt_oss_120b_fp4_grace_blackwell-gpt_oss_fp4_tp1_mtp0_8k1k]": 509.8764444444444, "perf/test_perf_sanity.py::test_e2e[aggr_upload-gpt_oss_120b_fp4_grace_blackwell-gpt_oss_fp4_tp2_1k8k]": 459.85290000000003, "perf/test_perf_sanity.py::test_e2e[aggr_upload-host_perf_deepseek_v3_lite-v3lite_fp8_bs8_128_256]": 596.7645321782178, - "perf/test_perf_sanity.py::test_e2e[aggr_upload-host_perf_llama8b-llama8b_fp16_bs8_128_256]": 257.1823105134474, "perf/test_perf_sanity.py::test_e2e[aggr_upload-qwen3_5_397b_fp4_blackwell-qwen3_5_397b_fp4_dep8_8k1k]": 698.3583333333333, "perf/test_perf_sanity.py::test_e2e[aggr_upload-qwen3_5_397b_fp4_blackwell-qwen3_5_397b_fp4_dep8_mtp3_8k1k]": 531.4513333333334, "perf/test_perf_sanity.py::test_e2e[aggr_upload-qwen3_5_397b_fp4_blackwell-qwen3_5_397b_fp4_tep4_mtp3_8k1k]": 404.2888333333333, @@ -968,8 +961,6 @@ "perf/test_visual_gen_perf_sanity.py::test_visual_gen_e2e[vg_upload-wan22_i2v_a14b_blackwell-wan22_i2v_a14b_nvfp4_trtllm_cfg2_ulysses4]": 342.01559999999995, "ray_orchestrator/RL/test_rl_perf_reproduce.py::test_rl_perf_reproduce[tp1_4instances]": 161.67143564356437, "ray_orchestrator/RL/test_rl_perf_reproduce.py::test_rl_perf_reproduce[tp2_2instances]": 103.54573267326732, - "stress_test/stress_test.py::test_run_stress_test[llama-v3-8b-instruct-hf_tp1-stress_time_300s_timeout_450s-GUARANTEED_NO_EVICT-pytorch-stress-test]": 710.0681999999999, - "stress_test/stress_test.py::test_run_stress_test[llama-v3-8b-instruct-hf_tp1-stress_time_300s_timeout_450s-MAX_UTILIZATION-pytorch-stress-test]": 702.3580999999999, "test_doc.py::test_url_validity": 49.48444444444444, "test_e2e.py::test_draft_token_tree_quickstart_advanced_eagle3[Llama-3.1-8b-Instruct-llama-3.1-model/Llama-3.1-8B-Instruct-EAGLE3-LLaMA3.1-Instruct-8B]": 54.35407142857143, "test_e2e.py::test_draft_token_tree_quickstart_advanced_eagle3_depth_1_tree[Llama-3.1-8b-Instruct-llama-3.1-model/Llama-3.1-8B-Instruct-EAGLE3-LLaMA3.1-Instruct-8B]": 55.181, @@ -1080,7 +1071,6 @@ "unittest/_torch/modeling -k \"modeling_llama\"": 116.07080075662043, "unittest/_torch/modeling -k \"modeling_mixtral\"": 72.60693384223919, "unittest/_torch/modeling -k \"modeling_nemotron_nano_v2_vl\"": 421.9780921658986, - "unittest/_torch/modeling -k \"modeling_nemotron_nas\"": 42.0892577092511, "unittest/_torch/modeling -k \"modeling_out_of_tree\"": 144.867929456112, "unittest/_torch/modeling -k \"modeling_phi3\"": 35.700971046770604, "unittest/_torch/modeling -k \"modeling_siglip\"": 204.05294545454547, @@ -1494,7 +1484,6 @@ "unittest/llmapi/test_llm_pytorch.py -m \"part1\"": 249.2021185086551, "unittest/llmapi/test_llm_pytorch.py -m \"part2\"": 349.5203164893617, "unittest/llmapi/test_llm_pytorch.py -m \"part3\"": 279.06796533333335, - "unittest/llmapi/test_llm_pytorch.py::test_nemotron_nas_lora": 194.5492, "unittest/llmapi/test_llm_quant.py": 21.754477644492912, "unittest/llmapi/test_llm_telemetry.py": 56.14217090620032, "unittest/llmapi/test_llm_telemetry.py::TestTelemetryArchitectureExtraction": 61.80126423690205, diff --git a/tests/integration/defs/.test_durations_aws_dfw b/tests/integration/defs/.test_durations_aws_dfw index 1b4de81b6b81..60962ab3784f 100644 --- a/tests/integration/defs/.test_durations_aws_dfw +++ b/tests/integration/defs/.test_durations_aws_dfw @@ -4,7 +4,6 @@ "accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_bfloat16_4gpus_python_scheduler[tp4-mtp_nextn=0]": 293.95560641004704, "accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_nvfp4_4gpus[moe_backend=CUTLASS-mtp_nextn=0-tp4-fp8kv=True-attention_dp=True-cuda_graph=True-overlap_scheduler=True-low_precision_combine=False-torch_compile=False]": 275.8909912491217, "accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_nvfp4_4gpus[moe_backend=CUTLASS-mtp_nextn=0-tp4-fp8kv=True-attention_dp=True-cuda_graph=True-overlap_scheduler=True-low_precision_combine=True-torch_compile=False]": 266.709816042101, - "accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp4_tp2pp2[torch_compile=True-enable_gemm_allreduce_fusion=False]": 908.2052957660053, "accuracy/test_llm_api_pytorch.py::TestNemotronV3Super::test_nvfp4_4gpus_online_eplb[moe_backend=CUTEDSL]": 641.9124943269417, "accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_bfloat16_4gpus_kv_cache_aware_routing[mtp_nextn=0]": 287.03202630905434, "accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_bfloat16_4gpus_kv_cache_aware_routing[mtp_nextn=2]": 350.3160223159939, @@ -15,8 +14,6 @@ "accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_nvfp4_4gpus[moe_backend=CUTLASS-mtp_nextn=0-tp2pp2-fp8kv=True-attention_dp=True-cuda_graph=True-overlap_scheduler=True-low_precision_combine=False-torch_compile=True]": 430.816315329168, "accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_w4_4gpus[v1_kv_cache-dp4-cutlass-auto]": 1067.7517247761134, "accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_w4_4gpus_online_eplb[fp8]": 813.0690455089789, - "accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp8_tp4[torch_compile=False]": 702.5942377618048, - "accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_nvfp4_tp4[torch_compile=False]": 848.7312017090153, "accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_bfloat16_4gpus_online_eplb[mtp_nextn=2-moe_backend=CUTLASS]": 413.68237058399245, "accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_nvfp4_4gpus[moe_backend=CUTLASS-mtp_nextn=0-ep4-fp8kv=True-attention_dp=True-cuda_graph=True-overlap_scheduler=True-low_precision_combine=False-torch_compile=True]": 319.0723833630327, "accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_nvfp4_4gpus[moe_backend=TRTLLM-mtp_nextn=0-ep4-fp8kv=True-attention_dp=True-cuda_graph=True-overlap_scheduler=True-low_precision_combine=False-torch_compile=False]": 332.56335388694424, diff --git a/tests/integration/defs/accuracy/references/SlimPajama-6B.yaml b/tests/integration/defs/accuracy/references/SlimPajama-6B.yaml deleted file mode 100644 index 421d0f38de81..000000000000 --- a/tests/integration/defs/accuracy/references/SlimPajama-6B.yaml +++ /dev/null @@ -1,2 +0,0 @@ -gradientai/Llama-3-8B-Instruct-Gradient-1048k: - - accuracy: 7.663 diff --git a/tests/integration/defs/accuracy/references/cnn_dailymail.yaml b/tests/integration/defs/accuracy/references/cnn_dailymail.yaml index 400f743ba0ce..672f44ea42db 100644 --- a/tests/integration/defs/accuracy/references/cnn_dailymail.yaml +++ b/tests/integration/defs/accuracy/references/cnn_dailymail.yaml @@ -34,12 +34,6 @@ gpt2-medium: accuracy: 22.249 gpt-next: - accuracy: 25.516 -microsoft/phi-2: - - accuracy: 31.255 -bigcode/starcoder2-7b: - - accuracy: 26.611 - - quant_algo: FP8 - accuracy: 26.611 state-spaces/mamba-130m-hf: - accuracy: 19.470 lmsys/vicuna-7b-v1.3: @@ -71,13 +65,6 @@ TinyLlama/TinyLlama-1.1B-Chat-v1.0: accuracy: 27.882 - extra_acc_spec: pp_size=4 accuracy: 15.123 -meta-llama/Meta-Llama-3-8B-Instruct: - - accuracy: 34.957 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 34.737 - - quant_algo: W8A16_GPTQ - accuracy: 34.858 meta-llama/Llama-3.1-8B: - accuracy: 24.360 - quant_algo: W8A8_SQ_PER_CHANNEL_PER_TOKEN_PLUGIN @@ -122,48 +109,6 @@ meta-llama/Llama-3.1-8B-Instruct: - quant_algo: FP8 extra_acc_spec: beam_width=2 accuracy: 31.201 -meta-llama/Llama-3.2-1B: - - accuracy: 27.427 - - quant_algo: W8A8_SQ_PER_CHANNEL_PER_TOKEN_PLUGIN - accuracy: 27.931 - - quant_algo: W8A8_SQ_PER_CHANNEL - accuracy: 25.631 - - quant_algo: W4A16_AWQ - accuracy: 25.028 - - quant_algo: W4A16_AWQ - kv_cache_quant_algo: INT8 - accuracy: 24.354 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 27.029 - - quant_algo: FP8 - accuracy: 27.029 - - quant_algo: FP8_PER_CHANNEL_PER_TOKEN - accuracy: 27.257 - - quant_algo: FP8_PER_CHANNEL_PER_TOKEN - extra_acc_spec: meta_recipe - accuracy: 27.614 - - extra_acc_spec: max_attention_window_size=960 - accuracy: 27.259 - - extra_acc_spec: max_attention_window_size=960;beam_width=4 - accuracy: 0 -meta-llama/Llama-3.2-3B: - - accuracy: 25.495 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 33.629 -meta-llama/Llama-3.3-70B-Instruct: - - quant_algo: FP8 - spec_dec_algo: Eagle - accuracy: 33.244 - - quant_algo: FP8 - spec_dec_algo: Eagle3 - accuracy: 33.244 - - quant_algo: NVFP4 - kv_cache_quant_algo: FP8 - accuracy: 34.383 - - quant_algo: FP8 - accuracy: 34.927 mistralai/Mistral-Small-3.1-24B-Instruct-2503: - accuracy: 29.20 - quant_algo: FP8 @@ -239,5 +184,3 @@ Qwen3/Qwen3-8B: - quant_algo: FP8_BLOCK_SCALES accuracy: 30 - accuracy: 30 -nvidia/Llama-3_3-Nemotron-Super-49B-v1: - - accuracy: 34.003 diff --git a/tests/integration/defs/accuracy/references/gpqa_diamond.yaml b/tests/integration/defs/accuracy/references/gpqa_diamond.yaml index 9483e8a43a40..d8adf5cd1c85 100644 --- a/tests/integration/defs/accuracy/references/gpqa_diamond.yaml +++ b/tests/integration/defs/accuracy/references/gpqa_diamond.yaml @@ -1,16 +1,3 @@ -meta-llama/Llama-3.3-70B-Instruct: - - accuracy: 45.96 - - quant_algo: NVFP4 - kv_cache_quant_algo: FP8 - accuracy: 45.55 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 48.03 - - quant_algo: FP8 - accuracy: 48.03 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 48.03 deepseek-ai/DeepSeek-R1: - quant_algo: NVFP4 accuracy: 70.45 @@ -35,30 +22,6 @@ deepseek-ai/DeepSeek-V3.2-Exp: - quant_algo: NVFP4 spec_dec_algo: MTP accuracy: 80.0 -nvidia/Llama-3_3-Nemotron-Super-49B-v1: - - accuracy: 44.95 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 42.42 - # GPQA diamond only contains 198 samples, so the score tends to have large variance. - # We repeated evaluation 7 times to choose a lower bound score for FP8, 42.42. - # random_seed=0: 47.98 - # random_seed=1: 42.42 - # random_seed=2: 52.02 - # random_seed=3: 51.52 - # random_seed=4: 48.48 - # random_seed=5: 47.47 - # random_seed=6: 45.96 -nvidia/Llama-3.1-Nemotron-Nano-8B-v1: - - accuracy: 40.40 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 39.39 -nvidia/Llama-3_1-Nemotron-Ultra-253B-v1: - - accuracy: 58.08 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 57.07 GPT-OSS/120B-MXFP4: - accuracy: 65.0 - spec_dec_algo: Eagle diff --git a/tests/integration/defs/accuracy/references/gsm8k.yaml b/tests/integration/defs/accuracy/references/gsm8k.yaml index 1a5d5adde3d5..1f21cc2a7cfe 100644 --- a/tests/integration/defs/accuracy/references/gsm8k.yaml +++ b/tests/integration/defs/accuracy/references/gsm8k.yaml @@ -33,24 +33,6 @@ meta-llama/Llama-3.1-8B-Instruct: - quant_algo: NVFP4 kv_cache_quant_algo: FP8 accuracy: 66.03 -meta-llama/Llama-3.3-70B-Instruct: - - accuracy: 83.78 - - quant_algo: NVFP4 - kv_cache_quant_algo: FP8 - accuracy: 87.33 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 90.30 - - quant_algo: FP8 - accuracy: 90.30 -meta-llama/Llama-4-Scout-17B-16E-Instruct: - - accuracy: 89.70 - - quant_algo: NVFP4 - kv_cache_quant_algo: FP8 - accuracy: 88.61 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 89.45 deepseek-ai/DeepSeek-V3-Lite: - accuracy: 64.74 - quant_algo: NVFP4 @@ -318,11 +300,6 @@ moonshotai/Kimi-K3: # SA spec dec is lossless; scores match the baseline within noise. - spec_dec_algo: SA accuracy: 96.5 -nvidia/Llama-3_3-Nemotron-Super-49B-v1: - - accuracy: 92.57 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 92.42 nvidia/Nemotron-MOE: - accuracy: 88.249 - quant_algo: FP8 @@ -331,16 +308,6 @@ nvidia/Nemotron-MOE: - quant_algo: NVFP4 kv_cache_quant_algo: FP8 accuracy: 63.268 -nvidia/Llama-3.1-Nemotron-Nano-8B-v1: - - accuracy: 37.15 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 28.39 -nvidia/Llama-3_1-Nemotron-Ultra-253B-v1: - - accuracy: 94.43 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 94.16 google/gemma-3-1b-it: - accuracy: 25.52 # score getting from lm-eval with HF implementation - quant_algo: FP8 @@ -433,8 +400,6 @@ GPT-OSS/20B-NVFP4: accuracy: 85.0 ByteDance-Seed/Seed-OSS-36B-Instruct: - accuracy: 90.8 -bigcode/starcoder2-7b: - - accuracy: 26.5 bigcode/starcoder2-15b: - accuracy: 54.5 poolside/laguna-XS.2: diff --git a/tests/integration/defs/accuracy/references/mmlu.yaml b/tests/integration/defs/accuracy/references/mmlu.yaml index b39f15a382f0..e092f23fca6d 100644 --- a/tests/integration/defs/accuracy/references/mmlu.yaml +++ b/tests/integration/defs/accuracy/references/mmlu.yaml @@ -1,8 +1,3 @@ -meta-llama/Meta-Llama-3-8B-Instruct: - - accuracy: 67.74 - - quant_algo: NVFP4 - kv_cache_quant_algo: FP8 - accuracy: 63.47 meta-llama/Llama-3.1-8B: - accuracy: 66.06 - quant_algo: NVFP4 @@ -35,59 +30,6 @@ meta-llama/Llama-3.1-8B-Instruct: - quant_algo: NVFP4 kv_cache_quant_algo: FP8 accuracy: 65.11 -meta-llama/Llama-3.2-1B: - - quant_algo: W8A8_SQ_PER_CHANNEL_PER_TOKEN_PLUGIN - accuracy: 32.72 - - quant_algo: W8A8_SQ_PER_CHANNEL - accuracy: 32.07 - - quant_algo: W4A16_AWQ - accuracy: 30.56 - - quant_algo: W4A16_AWQ - kv_cache_quant_algo: INT8 - accuracy: 31.29 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 31.02 - - quant_algo: FP8_PER_CHANNEL_PER_TOKEN - accuracy: 33.97 - - quant_algo: FP8_PER_CHANNEL_PER_TOKEN - extra_acc_spec: meta_recipe - accuracy: 33.87 - - extra_acc_spec: max_attention_window_size=960 - accuracy: 32.82 -meta-llama/Llama-3.2-3B: - - accuracy: 57.92 - - spec_dec_algo: Eagle3 - accuracy: 57.92 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 60.60 -meta-llama/Llama-3.3-70B-Instruct: - - accuracy: 81.31 - - spec_dec_algo: Eagle3 - accuracy: 81.31 - - quant_algo: FP8 - spec_dec_algo: Eagle - accuracy: 81.31 - - quant_algo: FP8 - spec_dec_algo: Eagle3 - accuracy: 81.31 - - quant_algo: NVFP4 - kv_cache_quant_algo: FP8 - accuracy: 78.78 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 80.40 - - quant_algo: FP8 - accuracy: 80.40 -meta-llama/Llama-4-Scout-17B-16E-Instruct: - - accuracy: 80.00 - - quant_algo: NVFP4 - kv_cache_quant_algo: FP8 - accuracy: 79.60 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 78.58 mistralai/Mistral-Small-3.1-24B-Instruct-2503: - accuracy: 81.7 - quant_algo: FP8 @@ -256,26 +198,7 @@ moonshotai/Kimi-K2.5: - quant_algo: NVFP4 kv_cache_quant_algo: FP8 accuracy: 89.67 -nvidia/Llama-3_3-Nemotron-Super-49B-v1: - - accuracy: 79.43 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 79.26 -nvidia/Llama-3.1-Nemotron-Nano-8B-v1: - - accuracy: 57.97 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 57.12 -bigcode/starcoder2-7b: - - accuracy: 41.35 - - quant_algo: FP8 - accuracy: 41.35 # TODO: update once https://nvbugs/5393849 is fixed. -nvidia/Llama-3_1-Nemotron-Ultra-253B-v1: - - accuracy: 83.70 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 83.36 mistralai/Ministral-8B-Instruct-2410: - accuracy: 66.35 - quant_algo: FP8 diff --git a/tests/integration/defs/accuracy/references/passkey_retrieval_128k.yaml b/tests/integration/defs/accuracy/references/passkey_retrieval_128k.yaml deleted file mode 100644 index f9f52a3ee867..000000000000 --- a/tests/integration/defs/accuracy/references/passkey_retrieval_128k.yaml +++ /dev/null @@ -1,2 +0,0 @@ -gradientai/Llama-3-8B-Instruct-Gradient-1048k: - - accuracy: 99 diff --git a/tests/integration/defs/accuracy/test_cli_flow.py b/tests/integration/defs/accuracy/test_cli_flow.py index cbae3409fc30..d3d4c47b0476 100644 --- a/tests/integration/defs/accuracy/test_cli_flow.py +++ b/tests/integration/defs/accuracy/test_cli_flow.py @@ -19,7 +19,7 @@ from ..conftest import (get_sm_version, llm_models_root, parametrize_with_ids, skip_no_nvls, skip_post_blackwell, skip_pre_ada, - skip_pre_blackwell, skip_pre_hopper) + skip_pre_hopper) from .accuracy_core import (MMLU, CliFlowAccuracyTestHarness, CnnDailymail, Humaneval, PassKeyRetrieval64k, ZeroScrolls) @@ -112,103 +112,6 @@ def test_fp8_prequantized(self, mocker): self.run(quant_algo=QuantAlgo.FP8, kv_cache_quant_algo=QuantAlgo.FP8) -# TODO: Remove the CLI tests once NIMs use PyTorch backend -@pytest.mark.timeout(5400) -class TestLlama3_3NemotronSuper49Bv1(CliFlowAccuracyTestHarness): - MODEL_NAME = "nvidia/Llama-3_3-Nemotron-Super-49B-v1" - MODEL_PATH = f"{llm_models_root()}/nemotron-nas/Llama-3_3-Nemotron-Super-49B-v1" - EXAMPLE_FOLDER = "models/core/nemotron_nas" - - @pytest.mark.skip_less_device(2) - @pytest.mark.skip_less_device_memory(80000) - def test_auto_dtype_tp2(self): - self.run(tasks=[MMLU(self.MODEL_NAME)], tp_size=2, dtype='auto') - - -class TestLlama3_1NemotronNano8Bv1(CliFlowAccuracyTestHarness): - MODEL_NAME = "nvidia/Llama-3.1-Nemotron-Nano-8B-v1" - MODEL_PATH = f"{llm_models_root()}/Llama-3.1-Nemotron-Nano-8B-v1" - EXAMPLE_FOLDER = "models/core/llama" - - def test_auto_dtype(self): - self.run(tasks=[MMLU(self.MODEL_NAME)], dtype='auto') - - @skip_pre_hopper - @pytest.mark.skip_device_not_contain(["H100", "H200", "B200"]) - def test_fp8_prequantized(self, mocker): - mocker.patch.object( - self.__class__, "MODEL_PATH", - f"{llm_models_root()}/Llama-3.1-Nemotron-Nano-8B-v1-FP8") - - self.run(tasks=[MMLU(self.MODEL_NAME)], - quant_algo=QuantAlgo.FP8, - kv_cache_quant_algo=QuantAlgo.FP8) - - -@pytest.mark.timeout(10800) -class TestNemotronUltra(CliFlowAccuracyTestHarness): - MODEL_NAME = "nvidia/Llama-3_1-Nemotron-Ultra-253B-v1" - MODEL_PATH = f"{llm_models_root()}/nemotron-nas/Llama-3_1-Nemotron-Ultra-253B-v1" - EXAMPLE_FOLDER = "models/core/nemotron_nas" - - @skip_pre_hopper - @pytest.mark.skip_less_device(8) - @pytest.mark.skip_less_device_memory(140000) - @parametrize_with_ids("cuda_graph", [False, True]) - @pytest.mark.parametrize("tp_size,pp_size", [(8, 1)], ids=["tp8"]) - def test_auto_dtype(self, cuda_graph, tp_size, pp_size): - extra_summarize_args = [] - if cuda_graph: - extra_summarize_args.append("--cuda_graph_mode") - - self.run(tasks=[MMLU(self.MODEL_NAME)], - tp_size=tp_size, - pp_size=pp_size, - extra_build_args=["--gemm_plugin=auto"], - extra_summarize_args=extra_summarize_args) - - @pytest.mark.skip( - reason="nemotron-nas scripts have to accommodate fp8 flags") - @skip_pre_hopper - @pytest.mark.skip_less_device(8) - @pytest.mark.skip_device_not_contain(["H100", "H200", "B200"]) - @parametrize_with_ids("cuda_graph", [False, True]) - @pytest.mark.parametrize("tp_size,pp_size", [(8, 1)], ids=["tp8"]) - def test_fp8_prequantized(self, cuda_graph, tp_size, pp_size, mocker): - mocker.patch.object( - self.__class__, "MODEL_PATH", - f"{llm_models_root()}/nemotron-nas/Llama-3_1-Nemotron-Ultra-253B-v1-FP8" - ) - - extra_summarize_args = [] - if cuda_graph: - extra_summarize_args.append("--cuda_graph_mode") - - self.run(tasks=[MMLU(self.MODEL_NAME)], - quant_algo=QuantAlgo.FP8, - kv_cache_quant_algo=QuantAlgo.FP8, - tp_size=tp_size, - pp_size=pp_size, - extra_build_args=["--gemm_plugin=auto"], - extra_summarize_args=extra_summarize_args) - - -@skip_post_blackwell -class TestPhi2(CliFlowAccuracyTestHarness): - MODEL_NAME = "microsoft/phi-2" - MODEL_PATH = f"{llm_models_root()}/phi-2" - EXAMPLE_FOLDER = "models/core/phi" - - @skip_post_blackwell - def test_auto_dtype(self): - self.run(dtype='auto') - - @skip_post_blackwell - @pytest.mark.skip_less_device(2) - def test_tp2(self): - self.run(tp_size=2) - - # Long sequence length test: # Model FP16 7B + 32K tokens in KV cache = 14 * 1024 MB + 32K * 0.5 MB = 30720 MB + scratch memory @pytest.mark.skip_less_device_memory(40000) @@ -327,55 +230,6 @@ def test_pp4(self): self.run(extra_acc_spec="pp_size=4", pp_size=4) -class TestLlama3_8BInstruct(CliFlowAccuracyTestHarness): - MODEL_NAME = "meta-llama/Meta-Llama-3-8B-Instruct" - MODEL_PATH = f"{llm_models_root()}/llama-models-v3/llama-v3-8b-instruct-hf" - EXAMPLE_FOLDER = "models/core/llama" - - def test_auto_dtype(self): - self.run(dtype='auto') - - @skip_pre_ada - def test_fp8(self): - self.run(quant_algo=QuantAlgo.FP8, kv_cache_quant_algo=QuantAlgo.FP8) - - @skip_pre_blackwell - def test_nvfp4(self): - self.run(tasks=[MMLU(self.MODEL_NAME)], - quant_algo=QuantAlgo.NVFP4, - kv_cache_quant_algo=QuantAlgo.FP8, - extra_build_args=["--gemm_plugin=disable"]) - - @pytest.mark.skip( - reason="Broken by modelopt. Will be fixed in next release") - @skip_pre_blackwell - @pytest.mark.parametrize("fuse_fp4_quant", [False, True], - ids=["disable_fused_quant", "enable_fused_quant"]) - @pytest.mark.parametrize( - "norm_quant_fusion", [False, True], - ids=["disable_norm_quant_fusion", "enable_norm_quant_fusion"]) - def test_nvfp4_gemm_plugin(self, fuse_fp4_quant: bool, - norm_quant_fusion: bool): - extra_build_args = ["--gemm_plugin=nvfp4"] - if fuse_fp4_quant: - extra_build_args.extend([ - "--use_paged_context_fmha=enable", - "--use_fp8_context_fmha=enable", "--fuse_fp4_quant=enable" - ]) - if norm_quant_fusion: - extra_build_args.append("--norm_quant_fusion=enable") - self.run(tasks=[MMLU(self.MODEL_NAME)], - quant_algo=QuantAlgo.NVFP4, - kv_cache_quant_algo=QuantAlgo.FP8, - extra_build_args=extra_build_args) - - -class TestLlama3_8BInstructGradient1048k(CliFlowAccuracyTestHarness): - MODEL_NAME = "gradientai/Llama-3-8B-Instruct-Gradient-1048k" - MODEL_PATH = f"{llm_models_root()}/llama-models-v3/Llama-3-8B-Instruct-Gradient-1048k" - EXAMPLE_FOLDER = "models/core/llama" - - class TestLlama3_1_8B(CliFlowAccuracyTestHarness): MODEL_NAME = "meta-llama/Llama-3.1-8B" MODEL_PATH = f"{llm_models_root()}/llama-3.1-model/Meta-Llama-3.1-8B" @@ -468,84 +322,6 @@ def test_medusa_fp8_prequantized(self, mocker): extra_summarize_args=extra_summarize_args) -class TestLlama3_2_1B(CliFlowAccuracyTestHarness): - MODEL_NAME = "meta-llama/Llama-3.2-1B" - MODEL_PATH = f"{llm_models_root()}/llama-3.2-models/Llama-3.2-1B" - EXAMPLE_FOLDER = "models/core/llama" - - def test_auto_dtype(self): - self.run(dtype='auto') - - @skip_pre_ada - def test_fp8(self): - self.run(quant_algo=QuantAlgo.FP8, kv_cache_quant_algo=QuantAlgo.FP8) - - @skip_pre_ada - @pytest.mark.skip_less_device(2) - @pytest.mark.parametrize( - "fp8_context_fmha", [False, True], - ids=["disable_fp8_context_fmha", "enable_fp8_context_fmha"]) - @pytest.mark.parametrize( - "reduce_fusion", [False, True], - ids=["disable_reduce_fusion", "enable_reduce_fusion"]) - def test_fp8_tp2(self, fp8_context_fmha: bool, reduce_fusion: bool): - if fp8_context_fmha: - extra_build_args = [ - "--use_fp8_context_fmha=enable", - "--use_paged_context_fmha=enable" - ] - else: - extra_build_args = [ - "--use_fp8_context_fmha=disable", - "--use_paged_context_fmha=disable" - ] - - if reduce_fusion: - extra_build_args.append("--reduce_fusion=enable") - else: - extra_build_args.append("--reduce_fusion=disable") - - self.run(quant_algo=QuantAlgo.FP8, - kv_cache_quant_algo=QuantAlgo.FP8, - tp_size=2, - extra_build_args=extra_build_args) - - @skip_pre_ada - @skip_post_blackwell - def test_fp8_rowwise(self): - self.run(quant_algo=QuantAlgo.FP8_PER_CHANNEL_PER_TOKEN) - - @skip_pre_ada - @skip_post_blackwell - def test_fp8_rowwise_meta_recipe(self): - self.run(quant_algo=QuantAlgo.FP8_PER_CHANNEL_PER_TOKEN, - extra_acc_spec="meta_recipe", - extra_convert_args=["--use_meta_fp8_rowwise_recipe"]) - - @pytest.mark.parametrize("max_gpu_percent", [0.1, 1.0]) - def test_weight_streaming(self, max_gpu_percent: float): - self.run(extra_build_args=["--weight_streaming"], - extra_summarize_args=["--gpu_weights_percent=0"]) - - for gpu_percent in [0.1, 0.5, 0.9, 1]: - if gpu_percent > max_gpu_percent: - break - self.extra_summarize_args = [f"--gpu_weights_percent={gpu_percent}"] - self.evaluate() - - -# TODO: Remove the CLI tests once NIMs use PyTorch backend -@pytest.mark.skip_less_device_memory(80000) -class TestLlama3_3_70BInstruct(CliFlowAccuracyTestHarness): - MODEL_NAME = "meta-llama/Llama-3.3-70B-Instruct" - MODEL_PATH = f"{llm_models_root()}/llama-3.3-models/Llama-3.3-70B-Instruct" - EXAMPLE_FOLDER = "models/core/llama" - - @pytest.mark.skip_less_device(8) - def test_auto_dtype_tp8(self): - self.run(tasks=[MMLU(self.MODEL_NAME)], tp_size=8, dtype='auto') - - class TestGemma2B(CliFlowAccuracyTestHarness): MODEL_NAME = "google/gemma-2b" MODEL_PATH = f"{llm_models_root()}/gemma/gemma-2b" diff --git a/tests/integration/defs/accuracy/test_llm_api_autodeploy.py b/tests/integration/defs/accuracy/test_llm_api_autodeploy.py index b52cc1e5f47e..1f2394634ea4 100644 --- a/tests/integration/defs/accuracy/test_llm_api_autodeploy.py +++ b/tests/integration/defs/accuracy/test_llm_api_autodeploy.py @@ -1442,15 +1442,6 @@ class TestModelRegistryAccuracy(LlmapiAccuracyTestHarness): id="google_gemma-3-1b-it"), pytest.param("mistralai/Ministral-8B-Instruct-2410", {}, [MMLU, GSM8K], id="mistralai_Ministral-8B-Instruct-2410"), - pytest.param("nvidia/Llama-3.1-Nemotron-Nano-8B-v1", {}, [MMLU, GSM8K], - id="nvidia_Llama-3.1-Nemotron-Nano-8B-v1"), - pytest.param( - "meta-llama/Llama-3.3-70B-Instruct", - {}, - [MMLU, GSM8K], - marks=(pytest.mark.skip_less_device_memory(80000), skip_pre_hopper), - id="meta-llama_Llama-3.3-70B-Instruct", - ), pytest.param( "deepseek-ai/DeepSeek-R1-0528", {}, diff --git a/tests/integration/defs/accuracy/test_llm_api_pytorch.py b/tests/integration/defs/accuracy/test_llm_api_pytorch.py index 8c5642626127..5744aedf8d64 100644 --- a/tests/integration/defs/accuracy/test_llm_api_pytorch.py +++ b/tests/integration/defs/accuracy/test_llm_api_pytorch.py @@ -1183,166 +1183,6 @@ def test_fp8_beam_search(self, enable_cuda_graph, enable_padding, extra_acc_spec="beam_width=2") -class TestLlama3_2_3B(LlmapiAccuracyTestHarness): - MODEL_NAME = "meta-llama/Llama-3.2-3B" - MODEL_PATH = f"{llm_models_root()}/llama-3.2-models/Llama-3.2-3B" - EXAMPLE_FOLDER = "models/core/llama" - - def test_auto_dtype(self): - with LLM(self.MODEL_PATH) as llm: - task = CnnDailymail(self.MODEL_NAME) - task.evaluate(llm) - task = MMLU(self.MODEL_NAME) - task.evaluate(llm) - - @skip_pre_hopper - def test_fp8_prequantized(self): - model_path = f"{llm_models_root()}/llama-3.2-models/Llama-3.2-3B-Instruct-FP8" - with LLM(model_path) as llm: - assert llm.args.quant_config.quant_algo == QuantAlgo.FP8 - task = CnnDailymail(self.MODEL_NAME) - task.evaluate(llm) - task = MMLU(self.MODEL_NAME) - task.evaluate(llm) - - -@pytest.mark.timeout(7200) -@pytest.mark.skip_less_device_memory(80000) -class TestLlama3_3_70BInstruct(LlmapiAccuracyTestHarness): - MODEL_NAME = "meta-llama/Llama-3.3-70B-Instruct" - - @pytest.mark.skip_less_mpi_world_size(8) - def test_auto_dtype_tp8(self): - model_path = f"{llm_models_root()}/llama-3.3-models/Llama-3.3-70B-Instruct" - with LLM(model_path, tensor_parallel_size=8) as llm: - task = MMLU(self.MODEL_NAME) - task.evaluate(llm) - task = GSM8K(self.MODEL_NAME) - task.evaluate(llm) - task = GPQADiamond(self.MODEL_NAME) - task.evaluate(llm, - extra_evaluator_kwargs=dict(apply_chat_template=True)) - - @pytest.mark.skip_less_mpi_world_size(2) - def test_auto_dtype_tp2(self): - _run_multinode_accuracy( - f"{llm_models_root()}/llama-3.3-models/Llama-3.3-70B-Instruct", - self.MODEL_NAME, - benchmarks=["mmlu"]) - - @skip_pre_hopper - @pytest.mark.skip_less_mpi_world_size(8) - @parametrize_with_ids("torch_compile", [False, True]) - @parametrize_with_ids("eagle3_one_model", [True, False]) - def test_fp8_eagle3_tp8(self, eagle3_one_model, torch_compile): - model_path = f"{llm_models_root()}/modelopt-hf-model-hub/Llama-3.3-70B-Instruct-fp8" - eagle_model_dir = f"{llm_models_root()}/EAGLE3-LLaMA3.3-Instruct-70B" - kv_cache_config = KvCacheConfig(free_gpu_memory_fraction=0.6) - spec_config = Eagle3DecodingConfig(max_draft_len=3, - speculative_model=eagle_model_dir, - eagle3_one_model=eagle3_one_model) - torch_compile_config = _get_default_torch_compile_config(torch_compile) - pytorch_config = dict( - disable_overlap_scheduler=not eagle3_one_model, - cuda_graph_config=CudaGraphConfig(max_batch_size=1), - torch_compile_config=torch_compile_config) - with LLM(model_path, - max_batch_size=16, - tensor_parallel_size=8, - speculative_config=spec_config, - kv_cache_config=kv_cache_config, - **pytorch_config) as llm: - task = CnnDailymail(self.MODEL_NAME) - task.evaluate(llm) - - @pytest.mark.skip_less_device(4) - @skip_pre_hopper - @parametrize_with_ids("torch_compile", [False, True]) - def test_fp8_tp4(self, torch_compile): - model_path = f"{llm_models_root()}/llama-3.3-models/Llama-3.3-70B-Instruct-FP8" - kv_cache_config = KvCacheConfig(free_gpu_memory_fraction=0.5) - torch_compile_config = _get_default_torch_compile_config(torch_compile) - with LLM(model_path, - tensor_parallel_size=4, - max_seq_len=8192, - max_batch_size=32, - kv_cache_config=kv_cache_config, - torch_compile_config=torch_compile_config) as llm: - assert llm.args.quant_config.quant_algo == QuantAlgo.FP8 - sampling_params = SamplingParams( - max_tokens=256, - temperature=0.0, - add_special_tokens=False, - ) - task = MMLU(self.MODEL_NAME) - task.evaluate(llm, sampling_params=sampling_params) - task = GSM8K(self.MODEL_NAME) - task.evaluate(llm, sampling_params=sampling_params) - task = GPQADiamond(self.MODEL_NAME) - task.evaluate(llm, - extra_evaluator_kwargs=dict(apply_chat_template=True)) - - @pytest.mark.skip_less_device(4) - @skip_pre_blackwell - @parametrize_with_ids("torch_compile", [False, True]) - def test_nvfp4_tp4(self, torch_compile): - model_path = f"{llm_models_root()}/llama-3.3-models/Llama-3.3-70B-Instruct-FP4" - kv_cache_config = KvCacheConfig(free_gpu_memory_fraction=0.5) - torch_compile_config = _get_default_torch_compile_config(torch_compile) - with LLM(model_path, - tensor_parallel_size=4, - max_batch_size=32, - kv_cache_config=kv_cache_config, - torch_compile_config=torch_compile_config) as llm: - assert llm.args.quant_config.quant_algo == QuantAlgo.NVFP4 - sampling_params = SamplingParams( - max_tokens=256, - temperature=0.0, - add_special_tokens=False, - ) - task = MMLU(self.MODEL_NAME) - task.evaluate(llm, sampling_params=sampling_params) - task = GSM8K(self.MODEL_NAME) - task.evaluate(llm, sampling_params=sampling_params) - task = GPQADiamond(self.MODEL_NAME) - task.evaluate(llm, - extra_evaluator_kwargs=dict(apply_chat_template=True)) - - @pytest.mark.skip_less_device(4) - @skip_pre_blackwell - @parametrize_with_ids("enable_gemm_allreduce_fusion", [False, True]) - @parametrize_with_ids("torch_compile", [False, True]) - def test_fp4_tp2pp2(self, enable_gemm_allreduce_fusion, torch_compile): - model_path = f"{llm_models_root()}/llama-3.3-models/Llama-3.3-70B-Instruct-FP4" - kv_cache_config = KvCacheConfig(free_gpu_memory_fraction=0.5) - torch_compile_config = _get_default_torch_compile_config(torch_compile) - - with (mock.patch.dict( - os.environ, { - "TRTLLM_GEMM_ALLREDUCE_FUSION_ENABLED": - str(int(enable_gemm_allreduce_fusion)) - }), - LLM(model_path, - tensor_parallel_size=2, - pipeline_parallel_size=2, - max_batch_size=32, - kv_cache_config=kv_cache_config, - torch_compile_config=torch_compile_config) as llm): - assert llm.args.quant_config.quant_algo == QuantAlgo.NVFP4 - sampling_params = SamplingParams( - max_tokens=256, - temperature=0.0, - add_special_tokens=False, - ) - task = MMLU(self.MODEL_NAME) - task.evaluate(llm, sampling_params=sampling_params) - task = GSM8K(self.MODEL_NAME) - task.evaluate(llm, sampling_params=sampling_params) - task = GPQADiamond(self.MODEL_NAME) - task.evaluate(llm, - extra_evaluator_kwargs=dict(apply_chat_template=True)) - - class TestMinistral8BInstruct(LlmapiAccuracyTestHarness): MODEL_NAME = "mistralai/Ministral-8B-Instruct-2410" MODEL_PATH = f"{llm_models_root()}/Ministral-8B-Instruct-2410" diff --git a/tests/integration/defs/conftest.py b/tests/integration/defs/conftest.py index b66464938258..0dffaf2de59e 100644 --- a/tests/integration/defs/conftest.py +++ b/tests/integration/defs/conftest.py @@ -674,23 +674,9 @@ def llama_v2_tokenizer_model_root(): def llama_model_root(request): models_root = llm_models_root() assert models_root, "Did you set LLM_MODELS_ROOT?" - if request.param == "llama-30b": - llama_model_root = os.path.join(models_root, "llama-models", - "llama-30b-hf") - elif request.param == "TinyLlama-1.1B-Chat-v1.0": + if request.param == "TinyLlama-1.1B-Chat-v1.0": llama_model_root = os.path.join(models_root, "llama-models-v2", "TinyLlama-1.1B-Chat-v1.0") - elif request.param == "llama-v3-8b-hf": - llama_model_root = os.path.join(models_root, "llama-models-v3", "8B") - elif request.param == "llama-v3-8b-instruct-hf": - llama_model_root = os.path.join(models_root, "llama-models-v3", - "llama-v3-8b-instruct-hf") - elif request.param == "Llama-3-8B-Instruct-Gradient-1048k": - llama_model_root = os.path.join(models_root, "llama-models-v3", - "Llama-3-8B-Instruct-Gradient-1048k") - elif request.param == "Llama-3-70B-Instruct-Gradient-1048k": - llama_model_root = os.path.join(models_root, "llama-models-v3", - "Llama-3-70B-Instruct-Gradient-1048k") elif request.param == "llama-3.1-8b": llama_model_root = os.path.join(models_root, "llama-3.1-model", "Meta-Llama-3.1-8B") @@ -703,21 +689,6 @@ def llama_model_root(request): elif request.param == "llama-3.1-8b-hf-nvfp4": llama_model_root = os.path.join(models_root, "nvfp4-quantized", "Meta-Llama-3.1-8B") - elif request.param == "llama-3.2-1b": - llama_model_root = os.path.join(models_root, "llama-3.2-models", - "Llama-3.2-1B") - elif request.param == "llama-3.2-1b-instruct": - llama_model_root = os.path.join(models_root, "llama-3.2-models", - "Llama-3.2-1B-Instruct") - elif request.param == "llama-3.2-3b": - llama_model_root = os.path.join(models_root, "llama-3.2-models", - "Llama-3.2-3B") - elif request.param == "llama-3.2-3b-instruct": - llama_model_root = os.path.join(models_root, "llama-3.2-models", - "Llama-3.2-3B-Instruct") - elif request.param == "llama-3.3-70b-instruct": - llama_model_root = os.path.join(models_root, "llama-3.3-models", - "Llama-3.3-70B-Instruct") assert os.path.exists( llama_model_root ), f"{llama_model_root} does not exist under NFS LLM_MODELS_ROOT dir" @@ -909,21 +880,6 @@ def mamba_model_root(request): return mamba_model_root -@pytest.fixture(scope="function") -def nemotron_nas_model_root(request): - models_root = llm_models_root() - assert models_root, "Did you set LLM_MODELS_ROOT?" - assert hasattr(request, "param"), "Param is missing!" - - nemotron_nas_model_root = os.path.join(models_root, "nemotron-nas", - request.param) - - assert exists( - nemotron_nas_model_root), f"{nemotron_nas_model_root} doesn't exist!" - - return nemotron_nas_model_root - - @pytest.fixture(scope="function") def llm_lora_model_root(request): "get lora model path" @@ -938,14 +894,7 @@ def llm_lora_model_root(request): model_list = [request.param] for item in model_list: - if item == "Japanese-Alpaca-LoRA-7b-v0": - model_root_list.append( - os.path.join(models_root, "llama-models", - "Japanese-Alpaca-LoRA-7b-v0")) - elif item == "luotuo-lora-7b-0.1": - model_root_list.append( - os.path.join(models_root, "llama-models", "luotuo-lora-7b-0.1")) - elif item == "peft-lora-starcoder2-15b-unity-copilot": + if item == "peft-lora-starcoder2-15b-unity-copilot": model_root_list.append( os.path.join( models_root, @@ -953,11 +902,6 @@ def llm_lora_model_root(request): "starcoder", "peft-lora-starcoder2-15b-unity-copilot", )) - elif item == "Llama-3_3-Nemotron-Super-49B-v1-lora-adapter_NIM_r32": - model_root_list.append( - os.path.join( - models_root, "nemotron-nas", - "Llama-3_3-Nemotron-Super-49B-v1-lora-adapter_NIM_r32")) elif item == "gpt-oss-20b-lora-adapter_NIM_r8": model_root_list.append( os.path.join(models_root, "gpt_oss", @@ -966,34 +910,6 @@ def llm_lora_model_root(request): return ",".join(model_root_list) -@pytest.fixture(scope="function") -def llm_dora_model_root(request): - "get dora model path" - models_root = llm_models_root() - assert models_root, "Did you set LLM_MODELS_ROOT?" - assert hasattr(request, "param"), "Param is missing!" - model_list = [] - model_root_list = [] - if isinstance(request.param, tuple): - model_list = list(request.param) - else: - model_list = [request.param] - - for item in model_list: - if item == "commonsense-llama-v3-8b-dora-r32": - model_root_list.append( - os.path.join( - models_root, - "llama-models-v3", - "DoRA-weights", - "llama_dora_commonsense_checkpoints", - "LLama3-8B", - "dora_r32", - )) - - return ",".join(model_root_list) - - @pytest.fixture(scope="module") @cached_in_llm_models_root("LongAlpaca-7B", True) def llm_long_alpaca_model_root(llm_venv): @@ -1005,37 +921,6 @@ def llm_long_alpaca_model_root(llm_venv): return long_alpaca_model_root -@pytest.fixture(scope="module") -@cached_in_llm_models_root("gpt-neox-20b", True) -def llm_gptneox_model_root(llm_venv): - "return gptneox model root" - - workspace = llm_venv.get_working_directory() - gptneox_model_root = os.path.join(workspace, "gpt-neox-20b") - - return gptneox_model_root - - -@pytest.fixture(scope="module") -@cached_in_llm_models_root("falcon-180b", True) -def llm_falcon_180b_model_root(): - "prepare falcon 180b model & return falcon model root" - raise RuntimeError("falcon 180b must be cached") - - -@pytest.fixture(scope="module") -@cached_in_llm_models_root("falcon-11B", True) -def llm_falcon_11b_model_root(llm_venv): - "prepare falcon-11B model & return falcon model root" - workspace = llm_venv.get_working_directory() - model_root = os.path.join(workspace, "falcon-11B") - - call(f"git clone https://huggingface.co/tiiuae/falcon-11B {model_root}", - shell=True) - - return model_root - - @pytest.fixture(scope="module") @cached_in_llm_models_root("email_composition", True) def llm_gpt2_next_8b_model_root(): diff --git a/tests/integration/defs/disaggregated/test_disaggregated.py b/tests/integration/defs/disaggregated/test_disaggregated.py index bf363a8ecfa5..3f71164b0094 100644 --- a/tests/integration/defs/disaggregated/test_disaggregated.py +++ b/tests/integration/defs/disaggregated/test_disaggregated.py @@ -2343,8 +2343,6 @@ def benchmark_model_root(request): model_path = os.path.join(models_root, "DeepSeek-V3-Lite", "fp8") elif (request.param == "DeepSeek-V3-Lite-bf16"): model_path = os.path.join(models_root, "DeepSeek-V3-Lite", "bf16") - elif request.param == "llama-v3-8b-hf": - model_path = os.path.join(models_root, "llama-models-v3", "8B") elif request.param == "llama-3.1-8b-instruct-hf-fp8": model_path = os.path.join(models_root, "llama-3.1-model", "Llama-3.1-8B-Instruct-FP8") diff --git a/tests/integration/defs/llmapi/test_llm_api_qa.py b/tests/integration/defs/llmapi/test_llm_api_qa.py index 532a0e355f45..0dec0d85b8d5 100644 --- a/tests/integration/defs/llmapi/test_llm_api_qa.py +++ b/tests/integration/defs/llmapi/test_llm_api_qa.py @@ -1,4 +1,19 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. # Confirm that the default backend is changed + import os from defs.common import venv_check_output @@ -7,8 +22,7 @@ model_path = os.path.join( llm_models_root(), - "llama-models-v3", - "llama-v3-8b-instruct-hf", + "Qwen3.5-4B", ) diff --git a/tests/integration/defs/perf/README_release_test.md b/tests/integration/defs/perf/README_release_test.md index 5555b0445ccc..35c5d34a28b9 100644 --- a/tests/integration/defs/perf/README_release_test.md +++ b/tests/integration/defs/perf/README_release_test.md @@ -186,12 +186,14 @@ cd tests/integration/defs ### 6.3 Add Test Case to Test List ```bash -echo "perf/test_perf.py::test_perf[llama_v3.3_70b_instruct_fp8-bench-pytorch-float8-input_output_len:128,128]" >> perf_test.txt +echo "perf/test_perf.py::test_perf[]" >> perf_test.txt ``` ### 6.4 Run Performance Test ```bash -pytest -v -s --test-prefix=H100_80GB_HBM3 --test-list=perf_test.txt -R=llama_v3.3_70b_instruct_fp8-bench-pytorch-float8-input_output_len:128,128 --output-dir=./output --perf --perf-log-formats=csv -o junit_logging=out-err +pytest -v -s --test-prefix=H100_80GB_HBM3 --test-list=perf_test.txt \ + -R= --output-dir=./output --perf \ + --perf-log-formats=csv -o junit_logging=out-err ``` ### 6.5 Command Parameters Explanation diff --git a/tests/integration/defs/perf/_model_paths.py b/tests/integration/defs/perf/_model_paths.py index dd52496fea99..c143aba4cfff 100644 --- a/tests/integration/defs/perf/_model_paths.py +++ b/tests/integration/defs/perf/_model_paths.py @@ -19,13 +19,6 @@ "llama_v3.1_8b_instruct": "llama-3.1-model/Llama-3.1-8B-Instruct", "llama_v3.1_8b_instruct_fp8": "llama-3.1-model/Llama-3.1-8B-Instruct-FP8", "llama_v3.1_8b_instruct_fp4": "modelopt-hf-model-hub/Llama-3.1-8B-Instruct-fp4", - "llama_v3.3_70b_instruct": "llama-3.3-models/Llama-3.3-70B-Instruct", - "llama_v3.3_70b_instruct_fp8": "modelopt-hf-model-hub/Llama-3.3-70B-Instruct-fp8", - "llama_v3.3_70b_instruct_fp4": "modelopt-hf-model-hub/Llama-3.3-70B-Instruct-fp4", - "llama_v3.3_nemotron_super_49b_v1.5_fp8": "nemotron-nas/Llama-3_3-Nemotron-Super-49B-v1_5-FP8", - "llama_v4_scout_17b_16e_instruct": "llama4-models/Llama-4-Scout-17B-16E-Instruct", - "llama_v4_scout_17b_16e_instruct_fp8": "llama4-models/Llama-4-Scout-17B-16E-Instruct-FP8", - "llama_v4_scout_17b_16e_instruct_fp4": "llama4-models/Llama-4-Scout-17B-16E-Instruct-FP4", "gemma_3_27b_it": "gemma/gemma-3-27b-it", "gemma_3_27b_it_fp8": "gemma/gemma-3-27b-it-fp8", "gemma_3_27b_it_fp4": "gemma/gemma-3-27b-it-FP4", diff --git a/tests/integration/defs/perf/allowed_configs.py b/tests/integration/defs/perf/allowed_configs.py index 223d7021bb02..52f16f30088c 100644 --- a/tests/integration/defs/perf/allowed_configs.py +++ b/tests/integration/defs/perf/allowed_configs.py @@ -115,510 +115,6 @@ class Config: _allowed_configs = { - "gpt_350m": - Config(name="gpt_350m", - family="gpt", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=256, - max_input_len=512, - max_seq_len=712, - ), - model_config=ModelConfig( - num_layers=24, - num_heads=16, - hidden_size=1024, - vocab_size=51200, - hidden_act='gelu', - n_positions=1024, - )), - "gpt_1.5b": - Config(name="gpt_1.5b", - family="gpt", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=256, - max_input_len=512, - max_seq_len=712, - ), - model_config=ModelConfig( - num_layers=48, - num_heads=25, - hidden_size=1600, - vocab_size=51200, - hidden_act='gelu', - n_positions=1024, - )), - "gpt_175b": - Config(name="gpt_175b", - family="gpt", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=64, - max_input_len=512, - max_seq_len=712, - ), - model_config=ModelConfig( - num_layers=96, - num_heads=96, - hidden_size=12288, - vocab_size=51200, - hidden_act='gelu', - n_positions=2048, - )), - "gpt_350m_moe": - Config(name="gpt_350m_moe", - family="gpt", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=256, - max_input_len=512, - max_seq_len=712, - ), - model_config=ModelConfig( - num_layers=24, - num_heads=16, - hidden_size=1024, - vocab_size=51200, - hidden_act='gelu', - n_positions=1024, - moe_num_experts=8, - moe_top_k=1, - )), - "gpt_350m_sq_per_tensor": - Config(name="gpt_350m_sq_per_tensor", - family="gpt", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=256, - max_input_len=512, - max_seq_len=712, - ), - model_config=ModelConfig( - num_layers=24, - num_heads=16, - hidden_size=1024, - vocab_size=51200, - hidden_act='gelu', - n_positions=1024, - quantization="int8_sq_per_tensor", - )), - "gpt_350m_sq_per_token_channel": - Config(name="gpt_350m_sq_per_token_channel", - family="gpt", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=256, - max_input_len=512, - max_seq_len=712, - ), - model_config=ModelConfig( - num_layers=24, - num_heads=16, - hidden_size=1024, - vocab_size=51200, - hidden_act='gelu', - n_positions=1024, - quantization="int8_sq_per_token_channel", - )), - "gpt_next_2b": - Config(name="gpt_next_2b", - family="gpt", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=256, - max_input_len=512, - max_seq_len=712, - ), - model_config=ModelConfig( - num_layers=24, - num_heads=16, - hidden_size=2048, - vocab_size=256000, - hidden_act='swiglu', - n_positions=1024, - position_embedding_type='rope_gpt_neox', - rotary_pct=0.5, - bias=False, - )), - "opt_350m": - Config(name="opt_350m", - family="opt", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=256, - max_input_len=512, - max_seq_len=712, - ), - model_config=ModelConfig( - num_layers=24, - num_heads=16, - hidden_size=1024, - vocab_size=50272, - hidden_act='relu', - n_positions=2048, - pre_norm=False, - do_layer_norm_before=False, - )), - "opt_2.7b": - Config(name="opt_2.7b", - family="opt", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=256, - max_input_len=512, - max_seq_len=712, - ), - model_config=ModelConfig( - num_layers=32, - num_heads=32, - hidden_size=2560, - vocab_size=50272, - hidden_act='relu', - n_positions=2048, - pre_norm=False, - do_layer_norm_before=True, - )), - "opt_6.7b": - Config(name="opt_6.7b", - family="opt", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=256, - max_input_len=512, - max_seq_len=712, - ), - model_config=ModelConfig( - num_layers=32, - num_heads=32, - hidden_size=4096, - vocab_size=50272, - hidden_act='relu', - n_positions=2048, - pre_norm=False, - do_layer_norm_before=True, - )), - "opt_30b": - Config(name="opt_30b", - family="opt", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=256, - max_input_len=512, - max_seq_len=712, - ), - model_config=ModelConfig( - num_layers=48, - num_heads=56, - hidden_size=7168, - vocab_size=50272, - hidden_act='relu', - n_positions=2048, - pre_norm=False, - do_layer_norm_before=True, - )), - "opt_66b": - Config(name="opt_66b", - family="opt", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=64, - max_input_len=512, - max_seq_len=712, - ), - model_config=ModelConfig( - num_layers=64, - num_heads=72, - hidden_size=9216, - vocab_size=50272, - hidden_act='relu', - n_positions=2048, - pre_norm=True, - do_layer_norm_before=True, - )), - "starcoder_15.5b": - Config(name="starcoder_15.5b", - family="gpt", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=256, - max_input_len=512, - max_seq_len=712, - ), - model_config=ModelConfig( - num_layers=40, - num_heads=48, - num_kv_heads=1, - hidden_size=6144, - vocab_size=49152, - hidden_act='gelu', - n_positions=8192, - )), - "llama_13b": - Config(name="llama_13b", - family="llama", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=128, - max_input_len=512, - max_seq_len=712, - ), - model_config=ModelConfig( - num_layers=40, - num_heads=40, - hidden_size=5120, - vocab_size=32000, - hidden_act='silu', - n_positions=4096, - inter_size=13824, - )), - "llama_30b": - Config(name="llama_30b", - family="llama", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=64, - max_input_len=512, - max_seq_len=712, - ), - model_config=ModelConfig( - num_layers=60, - num_heads=52, - hidden_size=6656, - vocab_size=32000, - hidden_act='silu', - n_positions=2048, - inter_size=17920, - )), - "llama_70b": - Config(name="llama_70b", - family="llama", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=64, - max_input_len=512, - max_seq_len=712, - ), - model_config=ModelConfig( - num_layers=80, - num_heads=64, - num_kv_heads=8, - hidden_size=8192, - vocab_size=32000, - hidden_act='silu', - n_positions=4096, - inter_size=28672, - )), - "llama_70b_long_context": - Config(name="llama_70b_long_context", - family="llama", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=16, - max_input_len=8000, - max_seq_len=8200, - ), - model_config=ModelConfig( - num_layers=80, - num_heads=64, - num_kv_heads=8, - hidden_size=8192, - vocab_size=32000, - hidden_act='silu', - n_positions=4096, - inter_size=28672, - multi_block_mode=True, - )), - "llama_70b_long_generation": - Config(name="llama_70b_long_generation", - family="llama", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=64, - max_input_len=200, - max_seq_len=16584, - ), - model_config=ModelConfig( - num_layers=80, - num_heads=64, - num_kv_heads=8, - hidden_size=8192, - vocab_size=32000, - hidden_act='silu', - n_positions=4096, - inter_size=28672, - multi_block_mode=True, - )), - "llama_70b_sq_per_tensor": - Config(name="llama_70b_sq_per_tensor", - family="llama", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=128, - max_input_len=512, - max_seq_len=712, - ), - model_config=ModelConfig( - num_layers=80, - num_heads=64, - num_kv_heads=8, - hidden_size=8192, - vocab_size=32000, - hidden_act='silu', - n_positions=4096, - inter_size=28672, - quantization="int8_sq_per_tensor", - )), - "gptj_6b": - Config(name="gptj_6b", - family="gptj", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=128, - max_input_len=512, - max_seq_len=712, - ), - model_config=ModelConfig( - num_layers=28, - num_heads=16, - hidden_size=4096, - vocab_size=50401, - hidden_act='gelu', - n_positions=1024, - rotary_dim=64, - )), - "gptneox_20b": - Config(name="gptneox_20b", - family="gptneox", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=16, - max_input_len=512, - max_seq_len=1024, - ), - model_config=ModelConfig( - num_layers=44, - num_heads=64, - hidden_size=6144, - vocab_size=50432, - hidden_act='gelu', - n_positions=2048, - rotary_dim=24, - )), - "chatglm_6b": - Config(name="chatglm_6b", - family="chatglm", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=256, - max_input_len=512, - max_seq_len=712, - ), - model_config=ModelConfig( - num_layers=28, - num_heads=32, - num_kv_heads=32, - hidden_size=4096, - inter_size=16384, - vocab_size=130528, - hidden_act='gelu', - n_positions=2048, - remove_input_padding=False, - )), - "chatglm2_6b": - Config(name="chatglm2_6b", - family="chatglm2", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=256, - max_input_len=512, - max_seq_len=712, - ), - model_config=ModelConfig( - num_layers=28, - num_heads=32, - num_kv_heads=2, - hidden_size=4096, - inter_size=13696, - vocab_size=65024, - hidden_act='swiglu', - n_positions=2048, - remove_input_padding=False, - )), - "chatglm3_6b": - Config(name="chatglm3_6b", - family="chatglm3", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=256, - max_input_len=512, - max_seq_len=712, - ), - model_config=ModelConfig( - num_layers=28, - num_heads=32, - num_kv_heads=2, - hidden_size=4096, - inter_size=13696, - vocab_size=65024, - hidden_act='swiglu', - n_positions=2048, - remove_input_padding=False, - )), - "glm_10b": - Config(name="glm_10b", - family="glm", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=128, - max_input_len=1024, - max_seq_len=1280, - ), - model_config=ModelConfig( - num_layers=48, - num_heads=64, - num_kv_heads=64, - hidden_size=4096, - inter_size=16384, - vocab_size=50304, - hidden_act='gelu', - n_positions=1024, - remove_input_padding=False, - )), - "bloom_560m": - Config(name="bloom_560m", - family="bloom", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=32, - max_input_len=1024, - max_seq_len=2048, - ), - model_config=ModelConfig( - num_layers=24, - num_heads=16, - hidden_size=1024, - vocab_size=250880, - hidden_act=None, - n_positions=2048, - )), - "bloom_176b": - Config(name="bloom_176b", - family="bloom", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=8, - max_input_len=1024, - max_seq_len=2048, - ), - model_config=ModelConfig( - num_layers=70, - num_heads=112, - hidden_size=14336, - vocab_size=250880, - hidden_act=None, - n_positions=2048, - )), "bert_base": Config(name="bert_base", family="bert", @@ -673,93 +169,6 @@ class Config: n_positions=1024, enable_context_fmha=False, )), - "falcon_rw_1b": - Config(name="falcon_rw_1b", - family="falcon", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=256, - max_input_len=1024, - max_seq_len=2048, - ), - model_config=ModelConfig( - num_layers=24, - num_heads=32, - hidden_size=2048, - vocab_size=50304, - hidden_act='gelu', - n_positions=2048, - bias=True, - use_alibi=True, - parallel_attention=False, - new_decoder_architecture=False, - )), - "falcon_7b": - Config(name="falcon_7b", - family="falcon", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=128, - max_input_len=512, - max_seq_len=712, - ), - model_config=ModelConfig( - num_layers=32, - num_heads=71, - num_kv_heads=1, - hidden_size=4544, - vocab_size=65024, - hidden_act='gelu', - n_positions=2048, - bias=False, - use_alibi=False, - parallel_attention=True, - new_decoder_architecture=False, - )), - "falcon_40b": - Config(name="falcon_40b", - family="falcon", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=64, - max_input_len=512, - max_seq_len=712, - ), - model_config=ModelConfig( - num_layers=60, - num_heads=128, - num_kv_heads=8, - hidden_size=8192, - vocab_size=65024, - hidden_act='gelu', - n_positions=2048, - bias=False, - use_alibi=False, - parallel_attention=True, - new_decoder_architecture=True, - )), - "falcon_180b": - Config(name="falcon_180b", - family="falcon", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=8, - max_input_len=1024, - max_seq_len=2048, - ), - model_config=ModelConfig( - num_layers=80, - num_heads=232, - num_kv_heads=8, - hidden_size=14848, - vocab_size=65024, - hidden_act='gelu', - n_positions=2048, - bias=False, - use_alibi=False, - parallel_attention=True, - new_decoder_architecture=True, - )), "t5_small": Config(name="t5_small", family="t5", @@ -1377,25 +786,6 @@ class Config: logits_soft_cap=30.0, state_dtype="float32", )), - "llama_v3_8b_instruct": - Config(name="llama_v3_8b_instruct", - family="llama", - benchmark_type="gpt", - build_config=BuildConfig( - max_batch_size=64, - max_input_len=1024, - max_seq_len=2048, - ), - model_config=ModelConfig( - num_layers=32, - num_heads=32, - num_kv_heads=8, - hidden_size=4096, - vocab_size=128256, - hidden_act='silu', - n_positions=8192, - inter_size=14336, - )) } diff --git a/tests/integration/defs/perf/host_perf/README.md b/tests/integration/defs/perf/host_perf/README.md index d54eee7c8f4c..8cf848c713c9 100644 --- a/tests/integration/defs/perf/host_perf/README.md +++ b/tests/integration/defs/perf/host_perf/README.md @@ -44,7 +44,7 @@ host-overhead-dominant YAML configs in `tests/scripts/perf-sanity/aggregated/hos ```bash # Run a specific host perf config through perf_sanity pytest tests/integration/defs/perf/test_perf_sanity.py -v \ - -k "host_perf_llama8b-llama8b_fp16_bs8_128_256" \ + -k "host_perf_deepseek_v3_lite-v3lite_fp8_bs8_128_256" \ --output-dir ./host_perf_results ``` diff --git a/tests/integration/defs/perf/pytorch_model_config.py b/tests/integration/defs/perf/pytorch_model_config.py index 373af821b377..07f825d55589 100644 --- a/tests/integration/defs/perf/pytorch_model_config.py +++ b/tests/integration/defs/perf/pytorch_model_config.py @@ -364,20 +364,6 @@ def get_model_yaml_config(model_label: str, }, } }, - # Llama-v4 Scout FP4 with cuda graph padding - { - 'patterns': ['llama_v4_scout_17b_16e_instruct_fp4'], - 'config': { - 'cuda_graph_config': { - 'enable_padding': - True, - 'batch_sizes': [ - 1, 2, 4, 8, 16, 32, 64, 128, 256, 384, 512, 1024, 2048, - 4096, 8192 - ] - } - } - }, # GPT-OSS 120B max throughput test { 'patterns': [ diff --git a/tests/integration/defs/perf/sampler_options_config.py b/tests/integration/defs/perf/sampler_options_config.py index de1824c8eced..ff75e6d559de 100644 --- a/tests/integration/defs/perf/sampler_options_config.py +++ b/tests/integration/defs/perf/sampler_options_config.py @@ -1,4 +1,4 @@ -# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 # # Licensed under the Apache License, Version 2.0 (the "License"); @@ -26,14 +26,4 @@ def get_sampler_options_config(model_label: str) -> dict: Returns: dict: sampler options config """ - # Labels are compared for equality, so they must be spelled exactly as - # PerfTestConfig.to_string() emits them: maxbs:/maxnt: are always injected - # and tp: is dropped when tp_size == num_gpus. - base_config = {} - if model_label in [ - 'llama_v3.3_70b_instruct_fp8-bench-pytorch-float8-maxbs:512-maxnt:2048-input_output_len:128,128-gpus:8', - ]: - base_config['top_k'] = 4 - base_config['top_p'] = 0.5 - base_config['temperature'] = 0.5 - return base_config + return {} diff --git a/tests/integration/defs/perf/test_perf.py b/tests/integration/defs/perf/test_perf.py index a62d445021f5..cf39d3e5fe72 100644 --- a/tests/integration/defs/perf/test_perf.py +++ b/tests/integration/defs/perf/test_perf.py @@ -839,12 +839,6 @@ def validate(self): [b >= 32 for b in self.batch_sizes] ), f"BERT with small BS is very unstable! Please increase to at least 32." - # GPT-350m and Bloom-560m with small BS are very unstable. Only run these small models with larger BS. - if self.model_name in ["gpt_350m", "bloom_560m"]: - assert all( - [b >= 32 for b in self.batch_sizes] - ), f"gpt_350m and bloom_560m with small BS are very unstable! Please increase to at least 32." - # Skip if not enough GPUs. TRTLLM_TOTAL_GPU_COUNT overrides # auto-detection for multi-node setups. total_gpus = int(os.environ["TRTLLM_TOTAL_GPU_COUNT"] diff --git a/tests/integration/defs/perf/test_perf_sanity.py b/tests/integration/defs/perf/test_perf_sanity.py index f99eaeee2535..59fba4ebcf96 100644 --- a/tests/integration/defs/perf/test_perf_sanity.py +++ b/tests/integration/defs/perf/test_perf_sanity.py @@ -38,15 +38,9 @@ from defs.trt_test_alternative import print_info from ..conftest import get_llm_root, llm_models_root -from ._model_paths import MODEL_PATH_DICT as _MODEL_PATH_DICT_BASE +from ._model_paths import MODEL_PATH_DICT from .perf_regression_utils import _percentile, process_and_upload_test_results -# Sanity-side path differs from test_perf for this key; preserve historical value. -MODEL_PATH_DICT = { - **_MODEL_PATH_DICT_BASE, - "llama_v3.3_70b_instruct_fp4": "llama-3.3-models/Llama-3.3-70B-Instruct-FP4", -} - SUPPORTED_GPU_MAPPING = { "GB200": "gb200", "GB300": "gb300", diff --git a/tests/integration/defs/stress_test/stress_test.py b/tests/integration/defs/stress_test/stress_test.py index c84d288336c4..46ff03b6a976 100644 --- a/tests/integration/defs/stress_test/stress_test.py +++ b/tests/integration/defs/stress_test/stress_test.py @@ -1,4 +1,4 @@ -# SPDX-FileCopyrightText: Copyright (c) 2022-2024 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-FileCopyrightText: Copyright (c) 2022-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 # # Licensed under the Apache License, Version 2.0 (the "License"); @@ -173,14 +173,14 @@ def request_count_stress_test(self) -> int: @dataclass(frozen=True) class PerformanceParams: """Dataclass to store test parameters for aiperf""" - input_len_mean: int = 64 # customized for tinyllama and llama-v3-8b-instruct-hf + input_len_mean: int = 64 # Customized for TinyLlama and Qwen3.5-4B. input_len_std: int = 16 - output_len_mean: int = 128 # customized for tinyllama and llama-v3-8b-instruct-hf + output_len_mean: int = 128 # Customized for TinyLlama and Qwen3.5-4B. output_len_std: int = 32 # test_timeout: # Maximum time allowed for the entire performance test to complete # Ensure indefinite runs specially for different concurrency values - test_timeout: int = 3600 # 1 hours for tinyllama and llama-v3-8b-instruct-hf + test_timeout: int = 3600 # One hour for TinyLlama and Qwen3.5-4B. concurrency_list: List[int] = field( default_factory=lambda: [8, 16, 32, 64, 128, 256]) @@ -434,10 +434,9 @@ def is_port_available(port: int, ModelConfig(model_dir="llama-models-v2/TinyLlama-1.1B-Chat-v1.0", tp_size=1, memory_requirement=12288), - # Configuration for Llama-v3 model + # Configuration for Qwen3.5-4B # memory_requirement is in MiB (12 GB = 12288 MiB) - ModelConfig(model_dir="llama-models-v3/llama-v3-8b-instruct-hf", - tp_size=1, + ModelConfig(model_dir="Qwen3.5-4B", tp_size=1, memory_requirement=12288), # Configuration for DeepSeek-V3 model # memory_requirement is in MiB (96 GB = 98304 MiB) diff --git a/tests/integration/defs/test_e2e.py b/tests/integration/defs/test_e2e.py index 13a58cfb71ef..6a7836b390c6 100644 --- a/tests/integration/defs/test_e2e.py +++ b/tests/integration/defs/test_e2e.py @@ -29,9 +29,8 @@ from .common import get_mmlu_accuracy, venv_check_call from .conftest import (get_device_count, get_sm_version, llm_models_root, - skip_no_sm120, skip_post_blackwell, skip_pre_ada, - skip_pre_blackwell, skip_pre_hopper, tests_path, - unittest_path) + skip_post_blackwell, skip_pre_ada, skip_pre_blackwell, + skip_pre_hopper, tests_path, unittest_path) sys.path.append(os.path.join(str(tests_path()), '/../examples/apps')) @@ -245,24 +244,6 @@ def parse_benchmark_output(self, output): return result -@pytest.mark.parametrize("model_name", ["meta-llama/Meta-Llama-3-8B-Instruct"], - ids=["llama3-8b"]) -@pytest.mark.parametrize("model_subdir", - ["llama-models-v3/llama-v3-8b-instruct-hf"], - ids=["llama-v3"]) -@pytest.mark.parametrize("use_pytorch_backend", [True], ids=["pytorch_backend"]) -def test_trtllm_bench_llmapi_launch(llm_root, llm_venv, model_name, - model_subdir, use_pytorch_backend): - runner = BenchRunner(llm_root=llm_root, - llm_venv=llm_venv, - model_name=model_name, - model_subdir=model_subdir, - streaming=False, - use_mpirun=True, - tp_size=2) - runner() - - @pytest.mark.parametrize( "model_name, llama_model_root", [pytest.param("TinyLlama-1.1B-Chat-v1.0", "TinyLlama-1.1B-Chat-v1.0")], @@ -897,12 +878,6 @@ def test_ptp_quickstart(llm_root, llm_venv): pytest.param('Llama3.1-8B-FP8', 'llama-3.1-model/Llama-3.1-8B-Instruct-FP8', marks=skip_pre_hopper), - pytest.param('Nemotron-Super-49B-v1-NVFP4', - 'nvfp4-quantized/Llama-3_3-Nemotron-Super-49B-v1_nvfp4_hf', - marks=skip_pre_hopper), - pytest.param('Nemotron-Super-49B-v1-FP8', - 'nemotron-nas/Llama-3_3-Nemotron-Super-49B-v1-FP8', - marks=skip_pre_hopper), pytest.param('Qwen3-30B-A3B', 'Qwen3/Qwen3-30B-A3B', marks=pytest.mark.skip_less_device_memory(80000)), @@ -914,16 +889,6 @@ def test_ptp_quickstart(llm_root, llm_venv): 'Qwen3-30B-A3B_nvfp4_hf', 'Qwen3/saved_models_Qwen3-30B-A3B_nvfp4_hf', marks=(skip_pre_blackwell, pytest.mark.skip_less_device_memory(20000))), - pytest.param( - 'Llama3.3-70B-FP8', - 'modelopt-hf-model-hub/Llama-3.3-70B-Instruct-fp8', - marks=(skip_pre_blackwell, pytest.mark.skip_less_device_memory(96000))), - pytest.param('Llama3.3-70B-FP4', - 'modelopt-hf-model-hub/Llama-3.3-70B-Instruct-fp4', - marks=skip_pre_blackwell), - pytest.param('Nemotron-Super-49B-v1-BF16', - 'nemotron-nas/Llama-3_3-Nemotron-Super-49B-v1', - marks=skip_pre_blackwell), pytest.param('GPT-OSS-20B', 'gpt_oss/gpt-oss-20b', marks=skip_pre_blackwell), pytest.param( @@ -948,12 +913,6 @@ def test_ptp_quickstart(llm_root, llm_venv): 'Qwen3/nvidia-Qwen3-32B-NVFP4', marks=skip_pre_blackwell), ("Qwen3-32B-bf16", "Qwen3/Qwen3-32B"), - pytest.param('Nemotron-Super-49B-v1.5-FP8', - 'nemotron-nas/Llama-3_3-Nemotron-Super-49B-v1_5-FP8', - marks=skip_pre_hopper), - pytest.param('Llama-4-Scout-17B-16E-FP4', - 'llama4-models/Llama-4-Scout-17B-16E-Instruct-FP4', - marks=skip_pre_blackwell), pytest.param('Nemotron-Nano-9B-v2-nvfp4', 'NVIDIA-Nemotron-Nano-9B-v2-NVFP4', marks=skip_pre_blackwell), @@ -984,10 +943,6 @@ def test_ptp_quickstart_advanced(llm_root, llm_venv, model_name, model_path): ] if "Qwen3" in model_name: cmds.append("--kv_cache_fraction=0.6") - if "Llama3.3-70B" in model_name: - cmds.append("--max_num_tokens=1024") - if "Llama-4" in model_name: - cmds.append("--max_seq_len=8192") llm_venv.run_cmd(cmds) @@ -1407,10 +1362,6 @@ def test_deepseek_r1_mtp_bench(llm_root, llm_venv): @pytest.mark.skip_less_device_memory(80000) @pytest.mark.parametrize("model_name,model_path,gpu_count", [ - pytest.param('Nemotron-Ultra-253B', - 'nemotron-nas/Llama-3_1-Nemotron-Ultra-253B-v1', - 8, - marks=(skip_pre_hopper, pytest.mark.timeout(12600))), pytest.param('DeepSeek-V3-671B-FP8', 'DeepSeek-V3-0324', 8, @@ -1423,13 +1374,7 @@ def test_ptp_quickstart_advanced_multi_gpus(llm_root, llm_venv, model_name, if gpu_count > get_device_count(): pytest.skip(f"Not enough GPUs for {model_name}") example_root = Path(os.path.join(llm_root, "examples", "llm-api")) - mapping = { - "Llama3.1-70B-BF16": 24.6, - "Llama3.1-70B-FP8": 58.5, - "Llama3.1-405B-FP8": 63.2, - "Nemotron-Ultra-253B": 72.3, - "DeepSeek-V3-671B-FP8": 83.8 - } + mapping = {"DeepSeek-V3-671B-FP8": 83.8} llm_venv.run_cmd([ str(example_root / "quickstart_advanced.py"), "--enable_chunked_prefill", @@ -1441,51 +1386,12 @@ def test_ptp_quickstart_advanced_multi_gpus(llm_root, llm_venv, model_name, ]) -@pytest.mark.skip_less_device_memory(80000) -@pytest.mark.parametrize("cuda_graph", [False, True]) -@pytest.mark.parametrize("tp_size, pp_size", [ - pytest.param(2, 2, marks=pytest.mark.skip_less_device(4)), - pytest.param(2, 4, marks=pytest.mark.skip_less_mpi_world_size(8)), -]) -@pytest.mark.parametrize("model_name,model_path", [ - pytest.param('Llama3.3-70B-FP8', - 'llama-3.3-models/Llama-3.3-70B-Instruct-FP8', - marks=skip_pre_hopper), -]) -def test_ptp_quickstart_advanced_pp_enabled(llm_root, llm_venv, model_name, - model_path, cuda_graph, tp_size, - pp_size): - print(f"Testing {model_name} on 8 GPUs.") - example_root = Path(os.path.join(llm_root, "examples", "llm-api")) - cmd = [ - str(example_root / "quickstart_advanced.py"), - "--enable_chunked_prefill", - "--model_dir", - f"{llm_models_root()}/{model_path}", - f"--tp_size={tp_size}", - f"--pp_size={pp_size}", - "--moe_ep_size=1", - "--kv_cache_fraction=0.6", - ] - if cuda_graph: - cmd.extend([ - "--use_cuda_graph", - "--cuda_graph_padding_enabled", - ]) - llm_venv.run_cmd(cmd) - - @skip_pre_hopper @pytest.mark.skip_less_mpi_world_size(8) @pytest.mark.parametrize("cuda_graph", [False, True]) @pytest.mark.parametrize("model_name,model_path", [ ("Llama-4-Maverick-17B-128E-Instruct-FP8", "llama4-models/nvidia/Llama-4-Maverick-17B-128E-Instruct-FP8"), - ("Llama-4-Scout-17B-16E-Instruct-FP8", - "llama4-models/Llama-4-Scout-17B-16E-Instruct-FP8"), - pytest.param('Llama-4-Scout-17B-16E-Instruct-FP4', - 'llama4-models/Llama-4-Scout-17B-16E-Instruct-FP4', - marks=skip_pre_blackwell), ]) def test_ptp_quickstart_advanced_8gpus_chunked_prefill_sq_22k( llm_root, llm_venv, model_name, model_path, cuda_graph): @@ -1509,30 +1415,6 @@ def test_ptp_quickstart_advanced_8gpus_chunked_prefill_sq_22k( llm_venv.run_cmd(cmd) -# This test is specifically to be run on 2 GPUs on Blackwell RTX 6000 Pro (SM120) architecture -# TODO: remove once we have a node with 8 GPUs and reuse test_ptp_quickstart_advanced_8gpus -@skip_no_sm120 -@pytest.mark.skip_less_device_memory(80000) -@pytest.mark.skip_less_device(2) -@pytest.mark.parametrize("model_name,model_path", [ - ('Nemotron-Super-49B-v1-BF16', - 'nemotron-nas/Llama-3_3-Nemotron-Super-49B-v1'), -]) -def test_ptp_quickstart_advanced_2gpus_sm120(llm_root, llm_venv, model_name, - model_path): - print(f"Testing {model_name} on 2 GPUs (SM120+).") - example_root = Path(os.path.join(llm_root, "examples", "llm-api")) - llm_venv.run_cmd([ - str(example_root / "quickstart_advanced.py"), - "--enable_chunked_prefill", - "--model_dir", - f"{llm_models_root()}/{model_path}", - "--tp_size=2", - "--max_num_tokens=256", - f"--kv_cache_fraction={_MEM_FRACTION_50}", - ]) - - @skip_pre_blackwell def test_ptp_quickstart_advanced_mixed_precision(llm_root, llm_venv): example_root = Path(os.path.join(llm_root, "examples", "llm-api")) @@ -1849,14 +1731,8 @@ def test_multi_nodes_eval(model_path: str, llm_api_config: Optional[dict[str, @pytest.mark.parametrize("tp_size,pp_size", [(2, 1), (1, 2)], ids=["tp2", "pp2"]) @pytest.mark.parametrize("model_path", [ - pytest.param('llama-3.3-models/Llama-3.3-70B-Instruct', - marks=skip_pre_hopper), pytest.param('Qwen3/saved_models_Qwen3-235B-A22B_nvfp4_hf', marks=skip_pre_blackwell), - pytest.param('llama4-models/Llama-4-Scout-17B-16E-Instruct-FP8', - marks=skip_pre_hopper), - pytest.param('llama4-models/Llama-4-Scout-17B-16E-Instruct', - marks=skip_pre_hopper), ]) def test_ptp_quickstart_advanced_multinode(llm_root, llm_venv, model_path, tp_size, pp_size): @@ -1898,8 +1774,6 @@ def test_ptp_quickstart_advanced_multinode(llm_root, llm_venv, model_path, @pytest.mark.skip_less_device_memory(80000) @skip_pre_hopper @pytest.mark.parametrize("model_dir,draft_model_dir", [ - ("modelopt-hf-model-hub/Llama-3.3-70B-Instruct-fp8", - "EAGLE3-LLaMA3.3-Instruct-70B"), ("Qwen3/Qwen3-30B-A3B", "Qwen3/Qwen3-30B-eagle3"), pytest.param("Qwen3/saved_models_Qwen3-235B-A22B_fp8_hf", "Qwen3/qwen3-235B-eagle3", diff --git a/tests/integration/defs/triton_server/conftest.py b/tests/integration/defs/triton_server/conftest.py index 75774eab9c9a..86fb46412bce 100644 --- a/tests/integration/defs/triton_server/conftest.py +++ b/tests/integration/defs/triton_server/conftest.py @@ -1,4 +1,18 @@ # -*- coding: utf-8 -*- +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. import datetime import os @@ -409,45 +423,6 @@ def mllama_model_root(): return mllama_model_root -@pytest.fixture(scope="session") -def llama_v3_8b_model_root(): - models_root = llm_models_root() - assert models_root, "Did you set LLM_MODELS_ROOT?" - llama_model_root = os.path.join(models_root, "llama-models-v3", - "llama-v3-8b-instruct-hf") - - assert os.path.exists( - llama_model_root - ), f"{llama_model_root} does not exist under NFS LLM_MODELS_ROOT dir" - return llama_model_root - - -@pytest.fixture(scope="session") -def llama3_v1_8b_model_root(): - models_root = llm_models_root() - assert models_root, "Did you set LLM_MODELS_ROOT?" - llama_model_root = os.path.join(models_root, "llama-3.1-model", - "Meta-Llama-3.1-8B") - - assert os.path.exists( - llama_model_root - ), f"{llama_model_root} does not exist under NFS LLM_MODELS_ROOT dir" - return llama_model_root - - -@pytest.fixture(scope="session") -def llama_v3_70b_model_root(): - models_root = llm_models_root() - assert models_root, "Did you set LLM_MODELS_ROOT?" - llama_model_root = os.path.join(models_root, "llama-models-v3", - "Llama-3-70B-Instruct-Gradient-1048k") - - assert os.path.exists( - llama_model_root - ), f"{llama_model_root} does not exist under NFS LLM_MODELS_ROOT dir" - return llama_model_root - - @pytest.fixture(scope="session") def vicuna_7b_model_root(): models_root = llm_models_root() diff --git a/tests/integration/test_lists/qa/llm_function_core.txt b/tests/integration/test_lists/qa/llm_function_core.txt index 8d611053b5d7..f7fd175d144e 100644 --- a/tests/integration/test_lists/qa/llm_function_core.txt +++ b/tests/integration/test_lists/qa/llm_function_core.txt @@ -587,19 +587,6 @@ accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_pard_sa_dynamic_ accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_pard_sa_global_pool accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_suffix_automaton_dynamic_draft_len accuracy/test_llm_api_pytorch.py::TestLlama3_1_8B_Instruct_RocketKV::test_auto_dtype -accuracy/test_llm_api_pytorch.py::TestLlama3_2_3B::test_auto_dtype -accuracy/test_llm_api_pytorch.py::TestLlama3_2_3B::test_fp8_prequantized -accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp4_tp2pp2[torch_compile=False-enable_gemm_allreduce_fusion=False] -accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp4_tp2pp2[torch_compile=False-enable_gemm_allreduce_fusion=True] -accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp4_tp2pp2[torch_compile=True-enable_gemm_allreduce_fusion=False] -accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp4_tp2pp2[torch_compile=True-enable_gemm_allreduce_fusion=True] -accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp8_eagle3_tp8[eagle3_one_model=False-torch_compile=False] -accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp8_eagle3_tp8[eagle3_one_model=True-torch_compile=False] -accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp8_eagle3_tp8[eagle3_one_model=True-torch_compile=True] -accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp8_tp4[torch_compile=False] -accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp8_tp4[torch_compile=True] -accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_nvfp4_tp4[torch_compile=False] -accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_nvfp4_tp4[torch_compile=True] accuracy/test_llm_api_pytorch.py::TestMiniMaxM2::test_4gpus[attention_dp=False-cuda_graph=True-overlap_scheduler=True-tp_size=4-ep_size=4] accuracy/test_llm_api_pytorch.py::TestMiniMaxM2_5::test_4gpus[attention_dp=False-cuda_graph=True-overlap_scheduler=True-tp_size=4-ep_size=4] accuracy/test_llm_api_pytorch.py::TestMiniMaxM3::test_auto_dtype[tp_size=8-ep_size=8] TIMEOUT (180) @@ -871,7 +858,6 @@ llmapi/test_llm_examples.py::test_llmapi_server_example test_e2e.py::test_eagle3_output_repetition_4gpus[Qwen3/Qwen3-30B-A3B-Qwen3/Qwen3-30B-eagle3] test_e2e.py::test_eagle3_output_repetition_4gpus[Qwen3/saved_models_Qwen3-235B-A22B_fp8_hf-Qwen3/qwen3-235B-eagle3] test_e2e.py::test_eagle3_output_repetition_4gpus[Qwen3/saved_models_Qwen3-235B-A22B_nvfp4_hf-Qwen3/qwen3-235B-eagle3] -test_e2e.py::test_eagle3_output_repetition_4gpus[modelopt-hf-model-hub/Llama-3.3-70B-Instruct-fp8-EAGLE3-LLaMA3.3-Instruct-70B] test_e2e.py::test_openai_chat_guided_decoding[openai/gpt-oss-120b] test_e2e.py::test_openai_chat_harmony_perf_metrics test_e2e.py::test_openai_kv_cache_contamination @@ -879,8 +865,6 @@ test_e2e.py::test_ptp_quickstart_advanced[Llama3.1-8B-BF16-llama-3.1-model/Meta- test_e2e.py::test_ptp_quickstart_advanced[Qwen3-30B-A3B-Qwen3/Qwen3-30B-A3B] test_e2e.py::test_ptp_quickstart_advanced_deepseek_r1_8gpus[DeepSeek-R1-DeepSeek-R1/DeepSeek-R1] test_e2e.py::test_ptp_quickstart_advanced_deepseek_r1_w4afp8_8gpus[DeepSeek-R1-W4AFP8-DeepSeek-R1/DeepSeek-R1-W4AFP8] -test_e2e.py::test_ptp_quickstart_advanced_pp_enabled[Llama3.3-70B-FP8-llama-3.3-models/Llama-3.3-70B-Instruct-FP8-2-2-False] -test_e2e.py::test_ptp_quickstart_advanced_pp_enabled[Llama3.3-70B-FP8-llama-3.3-models/Llama-3.3-70B-Instruct-FP8-2-4-True] test_e2e.py::test_ptp_quickstart_bert[TRTLLM-BertForSequenceClassification-bert/bert-base-uncased-yelp-polarity] test_e2e.py::test_ptp_quickstart_bert[VANILLA-BertForSequenceClassification-bert/bert-base-uncased-yelp-polarity] test_e2e.py::test_qwen_e2e_cpprunner_large_new_tokens[Qwen3/Qwen3-0.6B-Qwen3/Qwen3-0.6B] diff --git a/tests/integration/test_lists/qa/llm_spark_core.txt b/tests/integration/test_lists/qa/llm_spark_core.txt index 33851b1eb2be..e9250acc0a84 100644 --- a/tests/integration/test_lists/qa/llm_spark_core.txt +++ b/tests/integration/test_lists/qa/llm_spark_core.txt @@ -14,8 +14,6 @@ test_e2e.py::test_ptp_quickstart_advanced[Qwen3-32b-nvfp4-Qwen3/nvidia-Qwen3-32B test_e2e.py::test_ptp_quickstart_advanced[Nemotron-Nano-9B-v2-nvfp4-NVIDIA-Nemotron-Nano-9B-v2-NVFP4] test_e2e.py::test_ptp_quickstart_advanced[Qwen3-30B-A3B-Qwen3/Qwen3-30B-A3B] test_e2e.py::test_ptp_quickstart_advanced[Qwen3-30B-A3B_nvfp4_hf-Qwen3/saved_models_Qwen3-30B-A3B_nvfp4_hf] -test_e2e.py::test_ptp_quickstart_advanced[Llama3.3-70B-FP8-modelopt-hf-model-hub/Llama-3.3-70B-Instruct-fp8] -test_e2e.py::test_ptp_quickstart_advanced[Llama3.3-70B-FP4-modelopt-hf-model-hub/Llama-3.3-70B-Instruct-fp4] accuracy/test_llm_api_pytorch.py::TestLlama3_1_8B::test_auto_dtype accuracy/test_llm_api_pytorch.py::TestLlama3_1_8B::test_nvfp4 diff --git a/tests/integration/test_lists/qa/llm_spark_func.yml b/tests/integration/test_lists/qa/llm_spark_func.yml index b87ae1af0f99..f38f2fed1482 100644 --- a/tests/integration/test_lists/qa/llm_spark_func.yml +++ b/tests/integration/test_lists/qa/llm_spark_func.yml @@ -26,9 +26,6 @@ llm_spark_func: - test_e2e.py::test_ptp_quickstart_advanced[Qwen3-32b-nvfp4-Qwen3/nvidia-Qwen3-32B-NVFP4] - test_e2e.py::test_ptp_quickstart_advanced[Qwen3-30B-A3B-Qwen3/Qwen3-30B-A3B] - test_e2e.py::test_ptp_quickstart_advanced[Qwen3-30B-A3B_nvfp4_hf-Qwen3/saved_models_Qwen3-30B-A3B_nvfp4_hf] - - test_e2e.py::test_ptp_quickstart_advanced[Llama3.3-70B-FP8-modelopt-hf-model-hub/Llama-3.3-70B-Instruct-fp8] - - test_e2e.py::test_ptp_quickstart_advanced[Llama3.3-70B-FP4-modelopt-hf-model-hub/Llama-3.3-70B-Instruct-fp4] - - test_e2e.py::test_ptp_quickstart_advanced[Llama-4-Scout-17B-16E-FP4-llama4-models/Llama-4-Scout-17B-16E-Instruct-FP4] - test_e2e.py::test_ptp_quickstart_advanced[Nemotron-Nano-9B-v2-nvfp4-NVIDIA-Nemotron-Nano-9B-v2-NVFP4] - test_e2e.py::test_ptp_quickstart_advanced_eagle3[GPT-OSS-120B-Eagle3-gpt_oss/gpt-oss-120b-gpt_oss/gpt-oss-120b-Eagle3] - accuracy/test_llm_api_pytorch.py::TestLlama3_1_8B::test_auto_dtype @@ -58,10 +55,6 @@ llm_spark_func: gte: 2 lte: 2 tests: - - test_e2e.py::test_ptp_quickstart_advanced_multinode[llama-3.3-models/Llama-3.3-70B-Instruct-tp2] - test_e2e.py::test_ptp_quickstart_advanced_multinode[Qwen3/saved_models_Qwen3-235B-A22B_nvfp4_hf-tp2] - - test_e2e.py::test_ptp_quickstart_advanced_multinode[llama4-models/Llama-4-Scout-17B-16E-Instruct-FP8-tp2] - - test_e2e.py::test_ptp_quickstart_advanced_multinode[llama4-models/Llama-4-Scout-17B-16E-Instruct-tp2] - - accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_auto_dtype_tp2 - accuracy/test_llm_api_pytorch.py::TestQwen3_235B_A22B::test_nvfp4_2gpus[latency_moe_cutlass] - accuracy/test_llm_api_pytorch.py::TestQwen3_235B_A22B::test_nvfp4_2gpus[latency_moe_cutlass_eagle3] diff --git a/tests/integration/test_lists/qa/llm_spark_perf.yml b/tests/integration/test_lists/qa/llm_spark_perf.yml index 24975f87ff66..2214aa6122e2 100644 --- a/tests/integration/test_lists/qa/llm_spark_perf.yml +++ b/tests/integration/test_lists/qa/llm_spark_perf.yml @@ -34,12 +34,8 @@ llm_spark_perf: - perf/test_perf.py::test_perf[qwen3_14b_fp8-bench-pytorch-streaming-float8-maxbs:1-input_output_len:2048,128-reqs:1-con:1] - perf/test_perf.py::test_perf[qwen3_14b_fp4-bench-pytorch-streaming-float4-maxbs:1-input_output_len:2048,128-reqs:1-con:1] - perf/test_perf.py::test_perf[qwen3_14b-bench-pytorch-streaming-bfloat16-maxbs:1-input_output_len:2048,128-reqs:1-con:1] - - perf/test_perf.py::test_perf[llama_v3.3_70b_instruct_fp4-bench-pytorch-streaming-float4-maxbs:1-input_output_len:2048,128-reqs:1-con:1] - - perf/test_perf.py::test_perf[llama_v3.3_70b_instruct_fp8-bench-pytorch-streaming-float8-maxbs:1-input_output_len:2048,128-reqs:1-con:1] - perf/test_perf.py::test_perf[qwen3_30b_a3b_fp4-bench-pytorch-streaming-float4-maxbs:1-input_output_len:2048,128-reqs:1-con:1] - perf/test_perf.py::test_perf[qwen3_30b_a3b-bench-pytorch-streaming-bfloat16-maxbs:1-input_output_len:2048,128-reqs:1-con:1] - - perf/test_perf.py::test_perf[llama_v4_scout_17b_16e_instruct_fp4-bench-pytorch-streaming-float4-maxbs:1-input_output_len:2048,128-reqs:1-con:1] - - perf/test_perf.py::test_perf[llama_v3.3_nemotron_super_49b_v1.5_fp8-bench-pytorch-streaming-float8-maxbs:1-input_output_len:2048,128-reqs:1-con:1] - perf/test_perf.py::test_perf[qwen3_32b_fp4-bench-pytorch-streaming-float4-maxbs:1-input_output_len:2048,128-reqs:1-con:1] - perf/test_perf.py::test_perf[qwen3_32b-bench-pytorch-streaming-bfloat16-maxbs:1-input_output_len:2048,128-reqs:1-con:1] - perf/test_perf.py::test_perf[gemma_3_27b_it-bench-pytorch-streaming-bfloat16-maxbs:1-input_output_len:2048,128-reqs:1-con:1] @@ -57,11 +53,8 @@ llm_spark_perf: gte: 2 lte: 2 tests: - - perf/test_perf.py::test_perf[llama_v3.3_70b_instruct-bench-pytorch-streaming-bfloat16-maxbs:1-input_output_len:2048,128-reqs:1-con:1-tp:2-gpus:2] - perf/test_perf.py::test_perf[qwen3_235b_a22b_fp4-bench-pytorch-streaming-float4-maxbs:1-input_output_len:2048,128-kv_cache_dtype:fp8-reqs:1-con:1-tp:2-gpus:2] - perf/test_perf.py::test_perf[qwen3_235b_a22b_fp4-bench-pytorch-streaming-float4-maxbs:1-input_output_len:128,2048-kv_cache_dtype:fp8-reqs:1-con:1-tp:2-gpus:2] - - perf/test_perf.py::test_perf[llama_v4_scout_17b_16e_instruct_fp8-bench-pytorch-streaming-float8-maxbs:1-input_output_len:2048,128-kv_cache_dtype:fp8-reqs:1-con:1-ep:2-tp:2-gpus:2] - - perf/test_perf.py::test_perf[llama_v4_scout_17b_16e_instruct-bench-pytorch-streaming-bfloat16-maxbs:1-input_output_len:2048,128-reqs:1-con:1-tp:2-gpus:2] # Qwen3-235B-A22B-FP4 with Eagle3 speculative decoding - perf/test_perf.py::test_perf[qwen3_235b_a22b_fp4_eagle3-bench-pytorch-streaming-float4-maxbs:1-input_output_len:2048,128-kv_cache_dtype:fp8-reqs:1-con:1-tp:2-gpus:2] - perf/test_perf.py::test_perf[qwen3_235b_a22b_fp4_eagle3-bench-pytorch-streaming-float4-maxbs:1-input_output_len:128,2048-kv_cache_dtype:fp8-reqs:1-con:1-tp:2-gpus:2] diff --git a/tests/integration/test_lists/test-db/l0_a10.yml b/tests/integration/test_lists/test-db/l0_a10.yml index 0ff191759182..d7827dd72ec9 100644 --- a/tests/integration/test_lists/test-db/l0_a10.yml +++ b/tests/integration/test_lists/test-db/l0_a10.yml @@ -173,8 +173,6 @@ l0_a10: stage: post_merge backend: pytorch tests: - - stress_test/stress_test.py::test_run_stress_test[llama-v3-8b-instruct-hf_tp1-stress_time_300s_timeout_450s-GUARANTEED_NO_EVICT-pytorch-stress-test] - - stress_test/stress_test.py::test_run_stress_test[llama-v3-8b-instruct-hf_tp1-stress_time_300s_timeout_450s-MAX_UTILIZATION-pytorch-stress-test] - llmapi/test_llm_examples.py::test_llmapi_chat_example - llmapi/test_llm_examples.py::test_llmapi_server_example - llmapi/test_llm_examples.py::test_llmapi_kv_cache_connector[Qwen3/Qwen3-0.6B] diff --git a/tests/integration/test_lists/test-db/l0_a30.yml b/tests/integration/test_lists/test-db/l0_a30.yml index 57a27879065a..9060371dfca6 100644 --- a/tests/integration/test_lists/test-db/l0_a30.yml +++ b/tests/integration/test_lists/test-db/l0_a30.yml @@ -14,7 +14,6 @@ l0_a30: backend: pytorch tests: # ------------- PyTorch tests --------------- - - unittest/_torch/modeling -k "modeling_nemotron_nas" - unittest/_torch/modeling -k "modeling_qwen3" - unittest/_torch/modeling -k "modeling_out_of_tree" - unittest/_torch/modeling -k "modeling_speculative" diff --git a/tests/integration/test_lists/test-db/l0_b200.yml b/tests/integration/test_lists/test-db/l0_b200.yml index 40e7c99e3c59..d97eb7c521ff 100644 --- a/tests/integration/test_lists/test-db/l0_b200.yml +++ b/tests/integration/test_lists/test-db/l0_b200.yml @@ -303,9 +303,7 @@ l0_b200: - perf/host_perf/test_module_resource_manager.py::test_kv_cache_prepare_generation - perf/host_perf/test_module_resource_manager.py::test_kv_cache_prepare_context # ------------- Host perf E2E regression tests (reuse perf_sanity with host-overhead-dominant configs) --------------- - - perf/test_perf_sanity.py::test_e2e[aggr_upload-host_perf_llama8b-llama8b_fp16_bs8_128_256] - perf/test_perf_sanity.py::test_e2e[aggr_upload-host_perf_deepseek_v3_lite-v3lite_fp8_bs8_128_256] - - perf/test_perf_sanity.py::test_e2e[aggr_upload-host_perf_llama8b_spec_decode-llama8b_spec_bs1_128_128] - condition: ranges: system_gpu_count: diff --git a/tests/integration/test_lists/test-db/l0_dgx_b200.yml b/tests/integration/test_lists/test-db/l0_dgx_b200.yml index fb4bf93e7a9b..8709060bdbde 100644 --- a/tests/integration/test_lists/test-db/l0_dgx_b200.yml +++ b/tests/integration/test_lists/test-db/l0_dgx_b200.yml @@ -345,7 +345,6 @@ l0_dgx_b200: - accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_nvfp4_4gpus[moe_backend=CUTEDSL-mtp_nextn=2-ep4-fp8kv=True-attention_dp=True-cuda_graph=True-overlap_scheduler=True-low_precision_combine=False-torch_compile=False] - accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_nvfp4_4gpus[moe_backend=CUTEDSL-mtp_nextn=2-ep4-fp8kv=False-attention_dp=True-cuda_graph=False-overlap_scheduler=False-low_precision_combine=True-torch_compile=False] - accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_nvfp4_4gpus[moe_backend=CUTEDSL-mtp_nextn=2-ep4-fp8kv=True-attention_dp=True-cuda_graph=True-overlap_scheduler=True-low_precision_combine=True-torch_compile=False] - - accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp4_tp2pp2[torch_compile=False-enable_gemm_allreduce_fusion=False] - examples/visual_gen/test_visual_gen_wan.py::test_wan_t2v_example - examples/visual_gen/test_visual_gen_multi_gpu.py::test_wan22_t2v_lpips_against_golden_multi_gpu[ulysses4] - examples/visual_gen/test_visual_gen_multi_gpu.py::test_wan22_t2v_lpips_against_golden_multi_gpu[cfg2_ulysses2] diff --git a/tests/integration/test_lists/test-db/l0_dgx_h100.yml b/tests/integration/test_lists/test-db/l0_dgx_h100.yml index 43ab57a68f7a..7c671ca75747 100644 --- a/tests/integration/test_lists/test-db/l0_dgx_h100.yml +++ b/tests/integration/test_lists/test-db/l0_dgx_h100.yml @@ -101,7 +101,6 @@ l0_dgx_h100: - accuracy/test_llm_api_pytorch.py::TestNemotronV3Super::test_nvfp4_4gpus_hopper_w4a16 - test_e2e.py::test_ptp_quickstart_advanced_bs1 - test_e2e.py::test_ptp_quickstart_advanced_deepseek_v3_lite_4gpus_adp_balance[DeepSeek-V3-Lite-FP8-DeepSeek-V3-Lite/fp8] - - test_e2e.py::test_trtllm_bench_llmapi_launch[pytorch_backend-llama-v3-llama3-8b] # ------------- Disaggregated serving tests --------------- # Split test_py_cache_transceiver_mp.py by workflow (4 × 12 = 48 combos). # A single wrapper case ran ~3581 s on B200 and brushed the outer pytest diff --git a/tests/integration/test_lists/test-db/l0_dgx_h200.yml b/tests/integration/test_lists/test-db/l0_dgx_h200.yml index 395964dc693d..35a749f4c01c 100644 --- a/tests/integration/test_lists/test-db/l0_dgx_h200.yml +++ b/tests/integration/test_lists/test-db/l0_dgx_h200.yml @@ -45,7 +45,6 @@ l0_dgx_h200: - disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_ctxtp2ep2pp2_gentp4_one_mtp_block_reuse[DeepSeek-V3-Lite-fp8] - disaggregated/test_disaggregated.py::test_disaggregated_qwen3_32b_fp8[Qwen3/Qwen3-32B-FP8] - disaggregated/test_disaggregated.py::test_disaggregated_mixed_stress_test[req120-conc64-qwen3_32b_fp8_mixed_stress] - - unittest/llmapi/test_llm_pytorch.py::test_nemotron_nas_lora - accuracy/test_disaggregated_serving.py::TestDeepSeekV3Lite::test_gen_only_spec_dec - condition: ranges: @@ -136,7 +135,6 @@ l0_dgx_h200: - accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_fp8_4gpus[tp2pp2-fp8kv=True-attn_backend=FLASHINFER-torch_compile=True] - accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_fp8_4gpus[pp4-fp8kv=True-attn_backend=FLASHINFER-torch_compile=False] - accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_guided_decoding_4gpus[llguidance] - - test_e2e.py::test_trtllm_bench_llmapi_launch[pytorch_backend-llama-v3-llama3-8b] - test_e2e.py::test_trtllm_bench_mgmn - accuracy/test_llm_api_pytorch.py::TestQwen3_30B_A3B_Instruct_2507::test_skip_softmax_attention_4gpus[target_sparsity_0.5-fp8kv=False] - accuracy/test_llm_api_pytorch.py::TestQwen3_30B_A3B_Instruct_2507::test_skip_softmax_attention_4gpus[target_sparsity_0.5-fp8kv=True] diff --git a/tests/integration/test_lists/test-db/l0_gb200_multi_gpus.yml b/tests/integration/test_lists/test-db/l0_gb200_multi_gpus.yml index 3a474f887b87..12fe8abd2ec8 100644 --- a/tests/integration/test_lists/test-db/l0_gb200_multi_gpus.yml +++ b/tests/integration/test_lists/test-db/l0_gb200_multi_gpus.yml @@ -34,11 +34,6 @@ l0_gb200_multi_gpus: - accuracy/test_llm_api_pytorch.py::TestNemotronV3Ultra::test_nvfp4_4gpus_block_reuse[ADP4_MTP] - accuracy/test_llm_api_pytorch.py::TestQwen3_5_397B_A17B::test_nvfp4_4gpus_online_eplb[moe_backend=CUTEDSL] - accuracy/test_llm_api_pytorch.py::TestQwen3_5_397B_A17B::test_nvfp4_4gpus_online_eplb[moe_backend=TRTLLM] - - accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp8_tp4[torch_compile=False] - - accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp8_tp4[torch_compile=True] - - accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_nvfp4_tp4[torch_compile=False] - - accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_nvfp4_tp4[torch_compile=True] - - accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp4_tp2pp2[torch_compile=True-enable_gemm_allreduce_fusion=False] - accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_w4_4gpus[v2_kv_cache-tp4-trtllm-auto] - accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_w4_4gpus[v2_kv_cache-dp4-cutlass-auto] - accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_w4_4gpus_online_eplb[fp8] diff --git a/tests/integration/test_lists/waives.txt b/tests/integration/test_lists/waives.txt index c3a649190f0f..e4b41957e581 100644 --- a/tests/integration/test_lists/waives.txt +++ b/tests/integration/test_lists/waives.txt @@ -49,9 +49,6 @@ accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_w4_chunked_prefill[cutlass-au accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_bfloat16_4gpus[tp4-attn_backend=TRTLLM-torch_compile=False] SKIP (https://nvbugs/5616182) accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_fp8_4gpus[pp4-fp8kv=False-attn_backend=TRTLLM-torch_compile=False] SKIP (https://nvbugs/6437412) accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_fp8_4gpus[pp4-fp8kv=True-attn_backend=TRTLLM-torch_compile=False] SKIP (https://nvbugs/6278337) -accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp4_tp2pp2[torch_compile=False-enable_gemm_allreduce_fusion=False] SKIP (https://nvbugs/6428089) -accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp4_tp2pp2[torch_compile=True-enable_gemm_allreduce_fusion=True] SKIP (https://nvbugs/6211441) -accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp8_tp4[torch_compile=False] SKIP (https://nvbugs/6655987) accuracy/test_llm_api_pytorch.py::TestMiniMaxM3::test_nvfp4[use_msa=True] SKIP (https://nvbugs/6601633) accuracy/test_llm_api_pytorch.py::TestMistralLarge3_675B::test_nvfp4_4gpus[latency_moe_trtllm] SKIP (https://nvbugs/6572838) accuracy/test_llm_api_pytorch.py::TestMistralLarge3_675B::test_nvfp4_4gpus[latency_moe_trtllm_eagle] SKIP (https://nvbugs/6572838) @@ -217,7 +214,6 @@ full:H100/accuracy/test_llm_api_pytorch_multimodal.py::TestExaone4_5_33B::test_a full:H100/disaggregated/test_disaggregated.py::test_disaggregated_logprobs_serving[llama-3.1-8b-instruct] SKIP (https://nvbugs/6275959) full:H100/disaggregated/test_disaggregated.py::test_disaggregated_stress_test[input8k-output1k-conc512-qwen3_32b_fp8_stress] SKIP (https://nvbugs/6312828) full:H100_PCIe/unittest/auto_deploy/standalone/test_standalone_package.py::TestStandalonePackage::test_run_unit_tests SKIP (https://nvbugs/6672542) -full:H100_PCIe/unittest/llmapi/test_llm_pytorch.py::test_llama_7b_multi_lora_evict_and_reload_lora_gpu_cache SKIP (https://nvbugs/5682551) full:H20/accuracy/test_disaggregated_serving.py::TestDeepSeekV3Lite::test_auto_dtype[mtp_nextn=2-overlap_scheduler=False] SKIP (https://nvbugs/6345827) full:H20/accuracy/test_disaggregated_serving.py::TestDeepSeekV3Lite::test_auto_dtype[mtp_nextn=2-overlap_scheduler=True] SKIP (https://nvbugs/6345827) full:H20/accuracy/test_epd_disagg_multimodal.py::TestVideoMMEEPD::test_disaggregated_videomme[nemotron_nano_v3_omni_fp8] SKIP (https://nvbugs/6327718) diff --git a/tests/microbenchmarks/attention_perf/golden_attention.json b/tests/microbenchmarks/attention_perf/golden_attention.json index 29d42838434b..322111363846 100644 --- a/tests/microbenchmarks/attention_perf/golden_attention.json +++ b/tests/microbenchmarks/attention_perf/golden_attention.json @@ -62,7 +62,7 @@ "_src": "samesession build2801 REPEAT=5 B300-pcie: 0.0599/0.0532/0.0532/0.0553/0.0522; cv 5.04% (first val warmup outlier; drop-first cv 2.11%) -> gate ~20%, NEARLY USELESS. INSENSITIVE case: 0.053ms kernel too short for stable CUDA-event timing on fast B300. TODO raise warmup/iters or treat as discrete-only (launch_count=71 still guards structure)." } }, - "_comment_gqa": "Model-representative GQA cases (isolated harness, build2806). sm120 only so far; other arches bootstrap-skip until re-blessed on cluster. q8b=Qwen3-8B 32/8/128; l70b=64/8/128 (Qwen3-32B/Llama-3.3-70B/Nemotron-Super-49B). Continuous baselines are n=2 smoke values (1% floor binds since cv<0.1%); refine with a K=5 bootstrap.", + "_comment_gqa": "Model-representative GQA cases (isolated harness, build2806). sm120 only so far; other arches bootstrap-skip until re-blessed on cluster. q8b=Qwen3-8B 32/8/128; l70b=Qwen3-32B 64/8/128. Continuous baselines are n=2 smoke values (1% floor binds since cv<0.1%); refine with a K=5 bootstrap.", "attn_q8b_ctx_dispatch": { "NVIDIA RTX PRO 6000 Blackwell Server Edition": false }, diff --git a/tests/microbenchmarks/attention_perf/test_attention_perf_module.py b/tests/microbenchmarks/attention_perf/test_attention_perf_module.py index 117db1d252ae..2928762055f0 100644 --- a/tests/microbenchmarks/attention_perf/test_attention_perf_module.py +++ b/tests/microbenchmarks/attention_perf/test_attention_perf_module.py @@ -347,8 +347,7 @@ def test_attn_ctx_fmha_time(): # only on the HEAD SHAPE (num_heads, num_kv_heads, head_dim) + dtype + seq_len + # batch — so models that share a shape share a case: # q8b = Qwen3-8B -> 32 / 8 / 128 -# l70b = Qwen3-32B / Llama-3.3-70B / -# Llama-3.3-Nemotron-Super-49B -> 64 / 8 / 128 +# l70b = Qwen3-32B -> 64 / 8 / 128 # seq points map from perf ISL/OSL: prefill seq_len=2048 (long prompt), decode = # 1 new token over num_cached_tokens=1024 at batch 256 (throughput concurrency). # MLA (DeepSeek) and SWA+sink (gpt_oss) are separate attention families needing diff --git a/tests/microbenchmarks/qa/module_test_list.txt b/tests/microbenchmarks/qa/module_test_list.txt index b93651cf8b9d..73cb713e66e5 100644 --- a/tests/microbenchmarks/qa/module_test_list.txt +++ b/tests/microbenchmarks/qa/module_test_list.txt @@ -16,7 +16,7 @@ tests/microbenchmarks/attention_perf/test_attention_perf_module.py::test_attn_ct tests/microbenchmarks/attention_perf/test_attention_perf_module.py::test_attn_decode_xqa_time continuous # decode XQA gpu_time — the green-discrete + red-continuous blind-spot case # --- GQA model-representative shapes (ctx prefill seq=2048 / decode cached=1024 bs=256) --- -# q8b = Qwen3-8B (32/8/128); l70b = Qwen3-32B / Llama-3.3-70B / Nemotron-Super-49B (64/8/128) +# q8b = Qwen3-8B (32/8/128); l70b = Qwen3-32B (64/8/128) tests/microbenchmarks/attention_perf/test_attention_perf_module.py::test_gqa_ctx_dispatch[attn_q8b_ctx_dispatch] discrete tests/microbenchmarks/attention_perf/test_attention_perf_module.py::test_gqa_ctx_dispatch[attn_l70b_ctx_dispatch] discrete tests/microbenchmarks/attention_perf/test_attention_perf_module.py::test_gqa_decode_dispatch[attn_q8b_decode_dispatch] discrete diff --git a/tests/scripts/perf-sanity/aggregated/host_perf_llama8b.yaml b/tests/scripts/perf-sanity/aggregated/host_perf_llama8b.yaml deleted file mode 100644 index 7ce565adcb12..000000000000 --- a/tests/scripts/perf-sanity/aggregated/host_perf_llama8b.yaml +++ /dev/null @@ -1,83 +0,0 @@ -# Host performance regression test configs for Llama-3.1-8B (FP16). -# -# These configs are designed to be HOST-overhead-dominant: -# - Small batch sizes (1, 8) so GPU is not saturated -# - Short sequences (ISL=128, OSL=128-256) for high iteration rate -# - Single GPU (TP=1) to avoid communication overhead -# -# On these configs, host scheduling/sampling overhead is a significant fraction -# of per-iteration time, so host regressions show up in ITL/TPOT metrics. -# -# Llama-3.1-8B is a dense model that serves as the baseline for core -# scheduler/sampler host overhead (no MoE/MLA-specific paths). - -metadata: - model_name: llama_v3.1_8b_instruct - supported_gpus: - - H100 - - H200 - - B200 - - B300 -hardware: - gpus_per_node: 1 -server_configs: - # Config 1: Single-request decode — pure host overhead baseline - # With BS=1, each iteration has minimal GPU work. Host scheduling, - # sampling, and response handling overhead dominates ITL. - - name: "llama8b_fp16_bs1_128_128" - model_name: "llama_v3.1_8b_instruct" - tensor_parallel_size: 1 - max_batch_size: 1 - max_num_tokens: 256 - cuda_graph_config: - enable_padding: true - max_batch_size: 1 - kv_cache_config: - free_gpu_memory_fraction: 0.8 - client_configs: - - name: "con1_128_128" - concurrency: 1 - iterations: 200 - isl: 128 - osl: 128 - backend: "openai" - - # Config 2: Small batch decode — exercises scheduling with multiple requests - # BS=8 adds scheduling/batching overhead while keeping GPU work small. - - name: "llama8b_fp16_bs8_128_256" - model_name: "llama_v3.1_8b_instruct" - tensor_parallel_size: 1 - max_batch_size: 8 - max_num_tokens: 2048 - cuda_graph_config: - enable_padding: true - max_batch_size: 8 - kv_cache_config: - free_gpu_memory_fraction: 0.8 - client_configs: - - name: "con8_128_256" - concurrency: 8 - iterations: 200 - isl: 128 - osl: 256 - backend: "openai" - - # Config 3: Moderate batch decode — stresses scheduler, resource manager, - # and response handling at moderate concurrency while still host-visible. - - name: "llama8b_fp16_bs32_128_128" - model_name: "llama_v3.1_8b_instruct" - tensor_parallel_size: 1 - max_batch_size: 32 - max_num_tokens: 4096 - cuda_graph_config: - enable_padding: true - max_batch_size: 32 - kv_cache_config: - free_gpu_memory_fraction: 0.8 - client_configs: - - name: "con32_128_128" - concurrency: 32 - iterations: 200 - isl: 128 - osl: 128 - backend: "openai" diff --git a/tests/scripts/perf-sanity/aggregated/host_perf_llama8b_spec_decode.yaml b/tests/scripts/perf-sanity/aggregated/host_perf_llama8b_spec_decode.yaml deleted file mode 100644 index 513e5d01da40..000000000000 --- a/tests/scripts/perf-sanity/aggregated/host_perf_llama8b_spec_decode.yaml +++ /dev/null @@ -1,79 +0,0 @@ -# Host performance regression test configs for Llama-3.1-8B with speculative decoding. -# -# Speculative decoding adds significant host overhead through: -# - Draft token preparation (_prepare_draft_requests) -# - Draft model forward passes (drafter.prepare_draft_tokens) -# - Token verification and acceptance (_accept_draft_tokens) -# - KV cache rewind for rejected tokens -# -# These configs exercise the spec decode host paths at small batch sizes -# where host overhead dominates over GPU compute. -# -# REQUIREMENTS: -# - Target model: Llama-3.1-8B-Instruct (same as host_perf_llama8b.yaml) -# - Draft model: Llama-3.2-1B-Instruct (or equivalent small Llama model) -# Set in speculative_config.speculative_model path -# - Single GPU with enough memory for both models (~12GB + ~3GB) - -metadata: - model_name: llama_v3.1_8b_instruct - supported_gpus: - - H100 - - H200 - - B200 - - B300 -hardware: - gpus_per_node: 1 -server_configs: - # Config 1: Spec decode single request — baseline host overhead - # With BS=1, each iteration runs: draft → verify → accept/reject. - # The host overhead per iteration is ~2x non-spec-decode due to - # draft token management and verification. - - name: "llama8b_spec_bs1_128_128" - server_env_var: "TLLM_SPEC_DECODE_FORCE_NUM_ACCEPTED_TOKENS=3" - model_name: "llama_v3.1_8b_instruct" - tensor_parallel_size: 1 - max_batch_size: 4 - max_num_tokens: 1024 - speculative_config: - decoding_type: Draft_Target - max_draft_len: 3 - speculative_model: "llama-3.2-models/Llama-3.2-1B-Instruct" - cuda_graph_config: - enable_padding: true - max_batch_size: 4 - kv_cache_config: - free_gpu_memory_fraction: 0.8 - client_configs: - - name: "con1_128_128" - concurrency: 1 - iterations: 200 - isl: 128 - osl: 128 - backend: "openai" - - # Config 2: Spec decode small batch — exercises batched draft/verify - # BS=8 with spec decode means 8 requests each generating 3 draft tokens, - # stressing the scheduler, KV cache, and token verification paths. - - name: "llama8b_spec_bs8_128_128" - server_env_var: "TLLM_SPEC_DECODE_FORCE_NUM_ACCEPTED_TOKENS=3" - model_name: "llama_v3.1_8b_instruct" - tensor_parallel_size: 1 - max_batch_size: 16 - max_num_tokens: 4096 - speculative_config: - decoding_type: Draft_Target - max_draft_len: 3 - speculative_model: "llama-3.2-models/Llama-3.2-1B-Instruct" - cuda_graph_config: - enable_padding: true - max_batch_size: 16 - kv_cache_config: - free_gpu_memory_fraction: 0.8 - client_configs: - - name: "con8_128_128" - concurrency: 8 - iterations: 200 - isl: 128 - osl: 128 - backend: "openai" diff --git a/tests/scripts/perf-sanity/aggregated/llama_v3_3_70b_instruct_fp4_blackwell.yaml b/tests/scripts/perf-sanity/aggregated/llama_v3_3_70b_instruct_fp4_blackwell.yaml deleted file mode 100644 index ce97ab837d5e..000000000000 --- a/tests/scripts/perf-sanity/aggregated/llama_v3_3_70b_instruct_fp4_blackwell.yaml +++ /dev/null @@ -1,48 +0,0 @@ -metadata: - model_name: llama_v3.3_70b_instruct_fp4 - supported_gpus: - - B200 -hardware: - gpus_per_node: 8 -server_configs: - # TP4, ISL/OSL: 512/32 - - name: "llama70b_fp4_tp4_512_32" - model_name: "llama_v3.3_70b_instruct_fp4" - tensor_parallel_size: 4 - pipeline_parallel_size: 1 - max_batch_size: 512 - max_num_tokens: 2048 - cuda_graph_config: - enable_padding: true - max_batch_size: 512 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - client_configs: - - name: "con512_iter10_512_32" - concurrency: 512 - iterations: 10 - isl: 512 - osl: 32 - backend: "openai" - - # TP4, ISL/OSL: 1000/1000 - - name: "llama70b_fp4_tp4_1000_1000" - model_name: "llama_v3.3_70b_instruct_fp4" - tensor_parallel_size: 4 - pipeline_parallel_size: 1 - max_batch_size: 1024 - max_num_tokens: 4096 - cuda_graph_config: - enable_padding: true - max_batch_size: 1024 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - client_configs: - - name: "con512_iter10_1000_1000" - concurrency: 512 - iterations: 10 - isl: 1000 - osl: 1000 - backend: "openai" diff --git a/tests/test_common/llm_data.py b/tests/test_common/llm_data.py index 5cc8f6bf590f..eb07cdad1d87 100644 --- a/tests/test_common/llm_data.py +++ b/tests/test_common/llm_data.py @@ -32,7 +32,6 @@ "nvidia/Llama-3.1-8B-Instruct-FP8": "Llama-3.1-8B-Instruct-FP8", "nvidia/Llama-3.1-8B-Instruct-NVFP4": "Llama-3.1-8B-Instruct-NVFP4", "TinyLlama/TinyLlama-1.1B-Chat-v1.0": "llama-models-v2/TinyLlama-1.1B-Chat-v1.0", - "meta-llama/Llama-4-Scout-17B-16E-Instruct": "llama4-models/Llama-4-Scout-17B-16E-Instruct", "mistralai/Mistral-Small-3.1-24B-Instruct-2503": "Mistral-Small-3.1-24B-Instruct-2503", "Qwen/Qwen3-30B-A3B": "Qwen3/Qwen3-30B-A3B", "deepseek-ai/DeepSeek-V3": "DeepSeek-V3", @@ -58,9 +57,7 @@ "google/gemma-3n-E2B-it": "gemma/gemma-3n-E2B-it", "google/gemma-4-E2B-it": "gemma/gemma-4-E2B-it", "nvidia/Qwen3.5-397B-A17B-NVFP4": "Qwen3.5-397B-A17B-NVFP4", - "meta-llama/Llama-3.3-70B-Instruct": "llama-3.3-models/Llama-3.3-70B-Instruct", "mistralai/Ministral-8B-Instruct-2410": "Ministral-8B-Instruct-2410", - "nvidia/Llama-3.1-Nemotron-Nano-8B-v1": "Llama-3.1-Nemotron-Nano-8B-v1", "google/gemma-4-26B-A4B-it": "gemma/gemma-4-26B-A4B-it", "Qwen/Qwen3.5-35B-A3B": "Qwen3.5-35B-A3B", "Qwen/Qwen3.5-4B": "Qwen3.5-4B", diff --git a/tests/unittest/_torch/attention/model_attn_config.py b/tests/unittest/_torch/attention/model_attn_config.py index 0c844edfe58c..b3cd154f450d 100644 --- a/tests/unittest/_torch/attention/model_attn_config.py +++ b/tests/unittest/_torch/attention/model_attn_config.py @@ -226,34 +226,6 @@ class ModelAttnConfig: num_kv_heads=32, head_dim=128, ), - ModelAttnConfig( - "llama2_13b_mha", - "Llama-2-13B", - num_heads=40, - num_kv_heads=40, - head_dim=128, - ), - ModelAttnConfig( - "llama_30b_mha", - "Llama-30B", - num_heads=52, - num_kv_heads=52, - head_dim=128, - ), - ModelAttnConfig( - "llama_65b_mha", - "Llama-65B", - num_heads=64, - num_kv_heads=64, - head_dim=128, - ), - ModelAttnConfig( - "nemotron_nas_ultra_mha", - "Nemotron-NAS Ultra 253B", - num_heads=128, - num_kv_heads=128, - head_dim=128, - ), ModelAttnConfig( "generic_16h_mha_hd64", "Generic RoPE MHA backend shape retained for head-dimension coverage", @@ -282,13 +254,6 @@ class ModelAttnConfig: num_kv_heads=10, head_dim=128, ), - ModelAttnConfig( - "nemotron_nas_mha_hd64", - "Nemotron-NAS / DeciLM mini", - num_heads=32, - num_kv_heads=32, - head_dim=64, - ), ModelAttnConfig( "gpt_oss_gqa_hd64_gptj", "GPT-OSS full-attention layers (yarn, gptj-style rotation)", diff --git a/tests/unittest/_torch/modeling/test_modeling_nemotron_nas.py b/tests/unittest/_torch/modeling/test_modeling_nemotron_nas.py deleted file mode 100644 index 01bff76114b7..000000000000 --- a/tests/unittest/_torch/modeling/test_modeling_nemotron_nas.py +++ /dev/null @@ -1,525 +0,0 @@ -import unittest -from copy import deepcopy -from dataclasses import dataclass -from typing import Any - -import torch -import transformers -from parameterized import parameterized -from transformers import AutoConfig -from transformers.dynamic_module_utils import get_class_from_dynamic_module -from utils.llm_data import llm_models_root - -import tensorrt_llm -from tensorrt_llm._torch.attention_backend.utils import get_attention_backend -from tensorrt_llm._torch.metadata import KVCacheParams -from tensorrt_llm._torch.model_config import ModelConfig -from tensorrt_llm._torch.models.modeling_nemotron_nas import \ - NemotronNASForCausalLM -from tensorrt_llm._torch.pyexecutor.resource_manager import KVCacheManager -from tensorrt_llm.bindings.executor import KvCacheConfig -from tensorrt_llm.mapping import Mapping - -# Setup NEED_SETUP_CACHE_CLASSES_MAPPING to an empty dict for modeling_nemotron_nas.py -transformers.generation.utils.NEED_SETUP_CACHE_CLASSES_MAPPING = dict() - -NEMOTRON_NAS_MINI_CONFIG = { - "architectures": ["DeciLMForCausalLM"], - "attention_bias": - False, - "block_configs": [{ - "attention": { - "n_heads_in_group": 8, - "no_op": False, - "replace_with_linear": False - }, - "ffn": { - "ffn_mult": 2.0, - "no_op": False, - "replace_with_linear": False - } - }, { - "attention": { - "n_heads_in_group": 16, - "no_op": False, - "replace_with_linear": False - }, - "ffn": { - "ffn_mult": 2.0, - "no_op": False, - "replace_with_linear": False - } - }, { - "attention": { - "n_heads_in_group": None, - "no_op": False, - "replace_with_linear": True - }, - "ffn": { - "ffn_mult": 2.0, - "no_op": False, - "replace_with_linear": False - } - }, { - "attention": { - "n_heads_in_group": None, - "no_op": True, - "replace_with_linear": False - }, - "ffn": { - "ffn_mult": 2.0, - "no_op": False, - "replace_with_linear": False - } - }, { - "attention": { - "n_heads_in_group": 8, - "no_op": False, - "replace_with_linear": False - }, - "ffn": { - "ffn_mult": None, - "no_op": False, - "replace_with_linear": True - } - }, { - "attention": { - "n_heads_in_group": 4, - "no_op": False, - "replace_with_linear": False - }, - "ffn": { - "ffn_mult": None, - "no_op": True, - "replace_with_linear": False - } - }, { - "attention": { - "n_heads_in_group": 8, - "no_op": False, - "replace_with_linear": False - }, - "ffn": { - "ffn_mult": 2.0, - "no_op": False, - "replace_with_linear": False - } - }, { - "attention": { - "n_heads_in_group": 8, - "no_op": False, - "replace_with_linear": False - }, - "ffn": { - "ffn_mult": 2.0, - "no_op": False, - "replace_with_linear": False - } - }, { - "attention": { - "n_heads_in_group": 16, - "no_op": False, - "replace_with_linear": False - }, - "ffn": { - "ffn_mult": 2.0, - "no_op": False, - "replace_with_linear": False - } - }, { - "attention": { - "n_heads_in_group": None, - "no_op": False, - "replace_with_linear": True - }, - "ffn": { - "ffn_mult": 2.0, - "no_op": False, - "replace_with_linear": False - } - }, { - "attention": { - "n_heads_in_group": None, - "no_op": True, - "replace_with_linear": False - }, - "ffn": { - "ffn_mult": 2.0, - "no_op": False, - "replace_with_linear": False - } - }, { - "attention": { - "n_heads_in_group": 8, - "no_op": False, - "replace_with_linear": False - }, - "ffn": { - "ffn_mult": None, - "no_op": False, - "replace_with_linear": True - } - }, { - "attention": { - "n_heads_in_group": 4, - "no_op": False, - "replace_with_linear": False - }, - "ffn": { - "ffn_mult": None, - "no_op": True, - "replace_with_linear": False - } - }, { - "attention": { - "n_heads_in_group": 8, - "no_op": False, - "replace_with_linear": False - }, - "ffn": { - "ffn_mult": 2.0, - "no_op": False, - "replace_with_linear": False - } - }], - "bos_token_id": - 1, - "eos_token_id": - 2, - "hidden_act": - "silu", - "hidden_size": - 2048, - "initializer_range": - 0.02, - "intermediate_size": - None, - "max_position_embeddings": - 2048, - "model_type": - "deci", - "num_attention_heads": - 32, - "num_hidden_layers": - 14, - "num_key_value_heads": - None, - "rms_norm_eps": - 1e-06, - "rope_scaling": - None, - "rope_theta": - 10000.0, - "tie_word_embeddings": - False, - "torch_dtype": - "bfloat16", - "use_cache": - True, - "vocab_size": - 32128 -} - - -@dataclass(repr=False) -class Scenario: - backend: str - - def __repr__(self) -> str: - return f"backend:{self.backend.lower()}" - - -def reduce_nemotron_nas_config(mem_for_full_model: int, config_dict: dict[str, - Any]): - _, total_mem = torch.cuda.mem_get_info() - # scale model down if gpu memory is low - if total_mem < mem_for_full_model: - model_fraction = total_mem / mem_for_full_model - num_layers = int(config_dict["num_hidden_layers"] * model_fraction) - num_layers = min(num_layers, 32) - config_dict["num_hidden_layers"] = num_layers - config_dict["block_configs"] = config_dict["block_configs"][:num_layers] - - -class TestNemotronNAS(unittest.TestCase): - - def test_nemotron_nas_sanity(self): - config_dict = deepcopy(NEMOTRON_NAS_MINI_CONFIG) - # 8B * sizeof(float16) plus some extra for activations - mem_for_full_model = (2 + 1) * 8 * 2**(30) - reduce_nemotron_nas_config(mem_for_full_model, config_dict) - if config_dict["num_hidden_layers"] <= 0: - self.skipTest("Insufficient memory for a single NemotronNAS layer") - nemotron_nas_config = AutoConfig.from_pretrained( - llm_models_root() / "nemotron-nas/Llama-3_1-Nemotron-51B-Instruct", - trust_remote_code=True, - ) - nemotron_nas_config = nemotron_nas_config.from_dict(config_dict) - - dtype = nemotron_nas_config.torch_dtype - device = torch.device('cuda') - - model_config = ModelConfig(pretrained_config=nemotron_nas_config) - nemotron_nas = NemotronNASForCausalLM(model_config).to(dtype).to(device) - - input_ids = torch.tensor([100, 200, 300, 100, 200, 100, 400, 500], - dtype=torch.int, - device=device) - - num_blocks = 1000 - tokens_per_block = 128 - - if dtype == torch.half: - kv_cache_dtype = tensorrt_llm.bindings.DataType.HALF - elif dtype == torch.bfloat16: - kv_cache_dtype = tensorrt_llm.bindings.DataType.BF16 - else: - raise ValueError("Invalid dtype") - - mapping = Mapping(world_size=1, tp_size=1, rank=0) - kv_cache_config = KvCacheConfig(max_tokens=num_blocks * - tokens_per_block) - - num_layers = nemotron_nas.config.num_hidden_layers - num_kv_heads = nemotron_nas.config.num_key_value_heads - num_heads = nemotron_nas.config.num_attention_heads - head_dim = nemotron_nas.config.hidden_size // num_heads - max_seq_len = num_blocks * tokens_per_block - - context_sequence_lengths = [3, 2, 1] - sequence_lengths = context_sequence_lengths + [1, 1] - batch_size = len(sequence_lengths) - past_seen_tokens = [0, 0, 0, 62, 75] - request_ids = list(range(len(sequence_lengths))) - token_nums = (torch.tensor(past_seen_tokens) + - torch.tensor(sequence_lengths)).tolist() - prompt_lens = token_nums[:3] + past_seen_tokens[3:] - - kv_cache_manager = KVCacheManager( - kv_cache_config, - tensorrt_llm.bindings.internal.batch_manager.CacheType.SELF, - num_layers=num_layers, - num_kv_heads=num_kv_heads, - head_dim=head_dim, - tokens_per_block=tokens_per_block, - max_seq_len=max_seq_len, - max_batch_size=batch_size, - mapping=mapping, - dtype=kv_cache_dtype, - ) - kv_cache_manager.add_dummy_requests(request_ids, token_nums) - - metadata_cls = get_attention_backend(model_config.attn_backend).Metadata - attn_metadata = metadata_cls( - seq_lens=torch.tensor(sequence_lengths, dtype=torch.int), - num_contexts=len(context_sequence_lengths), - kv_cache_params=KVCacheParams( - use_cache=True, - num_cached_tokens_per_seq=past_seen_tokens, - ), - kv_cache_manager=kv_cache_manager, - request_ids=request_ids, - prompt_lens=prompt_lens, - max_num_requests=len(context_sequence_lengths) + 2, - max_num_tokens=8192, - ) - - position_ids = [] - for i, tokens in enumerate(past_seen_tokens): - seq_len = context_sequence_lengths[i] if i < len( - context_sequence_lengths) else 1 - position_id = torch.arange(tokens, - tokens + seq_len, - device=input_ids.device) - position_ids.append(position_id) - - position_ids = torch.cat(position_ids).unsqueeze(0) - - with torch.inference_mode(): - attn_metadata.prepare() - logits = nemotron_nas.forward(input_ids=input_ids, - position_ids=position_ids, - attn_metadata=attn_metadata) - - self.assertEqual(len(past_seen_tokens), logits.shape[0]) - - with torch.inference_mode(): - attn_metadata.prepare() - logits = nemotron_nas.forward(input_ids=input_ids, - position_ids=position_ids, - attn_metadata=attn_metadata, - return_context_logits=True) - self.assertEqual(input_ids.shape, logits.shape[:-1]) - - kv_cache_manager.shutdown() - - @parameterized.expand([ - Scenario(backend="VANILLA"), - Scenario(backend="FLASHINFER"), - Scenario(backend="TRTLLM"), - ], lambda testcase_func, param_num, param: - f"{testcase_func.__name__}[{param.args[0]}]") - @torch.no_grad() - @unittest.skip("https://nvbugspro.nvidia.com/bug/5439817") - def test_nemotron_nas_allclose_to_hf(self, scenario: Scenario) -> None: - """ - Compare output to HF - """ - backend = scenario.backend - metadata_cls = get_attention_backend(backend).Metadata - - torch.random.manual_seed(0) - config_dict = deepcopy(NEMOTRON_NAS_MINI_CONFIG) - # 8B * sizeof(float16) plus some extra for activations - # times 2, since we'll need 2 of these - mem_for_full_model = (2 + 1) * 8 * 2**(30) * 4 - reduce_nemotron_nas_config(mem_for_full_model, config_dict) - if config_dict["num_hidden_layers"] <= 0: - self.skipTest("Insufficient memory for a single NemotronNAS layer") - - nemotron_nas_ckpt = llm_models_root( - ) / "nemotron-nas/Llama-3_1-Nemotron-51B-Instruct" - nemotron_nas_config = AutoConfig.from_pretrained( - nemotron_nas_ckpt, - trust_remote_code=True, - ) - class_ref = nemotron_nas_config.auto_map["AutoModelForCausalLM"] - nemotron_nas_config = nemotron_nas_config.from_dict(config_dict) - dtype = nemotron_nas_config.torch_dtype - device = torch.device('cuda') - - model_class = get_class_from_dynamic_module(class_ref, - nemotron_nas_ckpt) - hf_nemotron_nas = model_class(nemotron_nas_config).to(dtype).to( - device).eval() - # This line populates the "variable" field in the NEED_SETUP_CACHE_CLASSES_MAPPING dict - hf_nemotron_nas._prepare_generation_config(None) - # And this line is the only way to access the only concrete Cache class DeciLMForCausalLM accepts - VariableCache = transformers.generation.utils.NEED_SETUP_CACHE_CLASSES_MAPPING[ - "variable"] - - model_config = ModelConfig(pretrained_config=nemotron_nas_config, - attn_backend=backend) - nemotron_nas = NemotronNASForCausalLM(model_config).to(dtype).to(device) - nemotron_nas.load_weights(hf_nemotron_nas.state_dict()) - - num_blocks = 1 - tokens_per_block = 128 - - kv_cache_config = KvCacheConfig(max_tokens=num_blocks * - tokens_per_block) - - num_layers = nemotron_nas.config.num_hidden_layers - num_kv_heads = nemotron_nas.config.num_key_value_heads - num_heads = nemotron_nas.config.num_attention_heads - head_dim = nemotron_nas.config.hidden_size // num_heads - max_seq_len = num_blocks * tokens_per_block - batch_size = 1 - - mapping = Mapping(world_size=1, tp_size=1, rank=0) - if dtype == torch.half: - kv_cache_dtype = tensorrt_llm.bindings.DataType.HALF - elif dtype == torch.bfloat16: - kv_cache_dtype = tensorrt_llm.bindings.DataType.BF16 - else: - raise ValueError("Invalid dtype") - - kv_cache_manager = KVCacheManager( - kv_cache_config, - tensorrt_llm.bindings.internal.batch_manager.CacheType.SELF, - num_layers=num_layers, - num_kv_heads=num_kv_heads, - head_dim=head_dim, - tokens_per_block=tokens_per_block, - max_seq_len=max_seq_len, - max_batch_size=batch_size, - mapping=mapping, - dtype=kv_cache_dtype, - ) - - # context - input_ids = torch.tensor([100, 200, 300, 100, 200, 100, 400, 500], - dtype=torch.int, - device=device) - - num_cached_tokens_per_seq = [0] - request_ids = [1] - token_nums = [input_ids.size(-1)] - prompt_lens = [input_ids.size(-1)] - kv_cache_manager.add_dummy_requests(request_ids, token_nums) - - attn_metadata = metadata_cls( - seq_lens=torch.tensor([input_ids.size(-1)], dtype=torch.int), - num_contexts=1, - kv_cache_params=KVCacheParams( - use_cache=True, - num_cached_tokens_per_seq=num_cached_tokens_per_seq, - ), - kv_cache_manager=kv_cache_manager, - request_ids=request_ids, - prompt_lens=prompt_lens, - max_num_requests=1, - max_num_tokens=8192, - ) - - position_ids = [torch.arange(0, input_ids.size(-1))] - position_ids = torch.cat(position_ids).unsqueeze(0).cuda() - # And, lastly, this is the simplest way of creating a Cache that `hf_nemotron_nas` will accept - past_key_values = VariableCache(config=nemotron_nas_config, - dtype=dtype, - batch_size=1) - with torch.inference_mode(): - attn_metadata.prepare() - logits = nemotron_nas.forward(input_ids=input_ids, - position_ids=position_ids, - attn_metadata=attn_metadata) - ref = hf_nemotron_nas.forward(input_ids=input_ids.unsqueeze(0), - position_ids=position_ids, - past_key_values=past_key_values, - use_cache=True) - - torch.testing.assert_close(logits, - ref.logits[:, -1].float(), - atol=0.1, - rtol=0.1) - - # gen - gen_input_ids = torch.tensor([600], dtype=torch.int, device=device) - - num_cached_tokens_per_seq = [input_ids.size(-1)] - - attn_metadata = metadata_cls( - seq_lens=torch.tensor([gen_input_ids.size(-1)], dtype=torch.int), - num_contexts=0, - kv_cache_params=KVCacheParams( - use_cache=True, - num_cached_tokens_per_seq=num_cached_tokens_per_seq, - ), - kv_cache_manager=kv_cache_manager, - request_ids=request_ids, - prompt_lens=prompt_lens, - max_num_requests=1, - max_num_tokens=8192, - ) - - gen_position_ids = [ - torch.arange(input_ids.size(-1), - input_ids.size(-1) + gen_input_ids.size(-1)) - ] - gen_position_ids = torch.cat(gen_position_ids).unsqueeze(0).cuda() - with torch.inference_mode(): - attn_metadata.prepare() - logits = nemotron_nas.forward(input_ids=gen_input_ids, - position_ids=gen_position_ids, - attn_metadata=attn_metadata) - ref = hf_nemotron_nas.forward(input_ids=gen_input_ids.unsqueeze(0), - position_ids=gen_position_ids, - past_key_values=ref.past_key_values, - use_cache=True) - - torch.testing.assert_close(logits, - ref.logits[:, -1].float(), - atol=0.1, - rtol=0.1) - - kv_cache_manager.shutdown() diff --git a/tests/unittest/_torch/speculative/hw_agnostic/test_draft_len_schedule.py b/tests/unittest/_torch/speculative/hw_agnostic/test_draft_len_schedule.py index 67e5514cb9c2..acf85897ae49 100644 --- a/tests/unittest/_torch/speculative/hw_agnostic/test_draft_len_schedule.py +++ b/tests/unittest/_torch/speculative/hw_agnostic/test_draft_len_schedule.py @@ -1,3 +1,18 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + import os import sys @@ -5,6 +20,7 @@ import torch from tensorrt_llm import LLM, SamplingParams +from tensorrt_llm._torch.speculative.utils import get_draft_len_for_batch_size from tensorrt_llm.llmapi import DraftTargetDecodingConfig, KvCacheConfig, NGramDecodingConfig sys.path.append(os.path.join(os.path.dirname(__file__), "..")) @@ -29,18 +45,21 @@ def enforce_single_worker(monkeypatch): "drafter_type,schedule", [ ("ngram", {1: 3, 4: 2, 8: 1}), + ("model_drafter", {1: 3, 4: 2, 8: 1}), ], ) @pytest.mark.high_cuda_memory -def test_correctness_across_batch_sizes(drafter_type: str, schedule: dict): +def test_correctness_across_batch_sizes( + enforce_single_worker, monkeypatch, drafter_type: str, schedule: dict +): total_mem_gb = torch.cuda.get_device_properties(0).total_memory / 1e9 memory_required = 30 if drafter_type == "model_drafter" else 20 if total_mem_gb < memory_required: pytest.skip(f"Not enough memory (need {memory_required}GB, have {total_mem_gb:.1f}GB)") models_path = llm_models_root() - target_model = f"{models_path}/llama-3.1-model/Llama-3.1-8B-Instruct" - draft_model = f"{models_path}/llama-3.2-models/Llama-3.2-3B-Instruct" + target_model = f"{models_path}/Qwen3/Qwen3-8B" + draft_model = f"{models_path}/Qwen3/Qwen3-0.6B" max_batch_size = 8 max_draft_len = max(schedule.values()) # Use max from schedule @@ -99,28 +118,72 @@ def test_correctness_across_batch_sizes(drafter_type: str, schedule: dict): ) for i in range(len(prompts)) ] + + if drafter_type == "model_drafter": + prompts = ["The capital of France is"] * max_batch_size + sampling_params_list = [ + SamplingParams( + max_tokens=max_tokens, + temperature=0, + seed=42, + ignore_eos=True, + top_k=1, + top_p=1.0, + ) + for max_tokens in [4] * 4 + [8] * 3 + [12] + ] + # With dynamic draft_len_schedule llm_with_schedule = LLM(**llm_common_config, speculative_config=spec_config) + + runtime_schedule = [] + if drafter_type == "model_drafter": + executor = llm_with_schedule._executor.engine + original_handle_dynamic_draft_len = executor._handle_dynamic_draft_len + + def instrumented_handle_dynamic_draft_len(scheduled_batch): + original_handle_dynamic_draft_len(scheduled_batch) + runtime_schedule.append( + (scheduled_batch.batch_size, executor.model_engine.runtime_draft_len) + ) + + monkeypatch.setattr( + executor, "_handle_dynamic_draft_len", instrumented_handle_dynamic_draft_len + ) + results_with_schedule = llm_with_schedule.generate(prompts, sampling_params_list) generated_text_with_schedule = [result.outputs[0].text for result in results_with_schedule] llm_with_schedule.shutdown() - # Reference: spec decode with fixed max_draft_len (no schedule) - if drafter_type == "ngram": - spec_config_fixed = NGramDecodingConfig( - max_draft_len=max_draft_len, - max_matching_ngram_size=2, - draft_len_schedule=None, # No schedule - fixed draft length - is_keep_all=True, - is_use_oldest=True, - is_public_pool=False, - ) - else: - # skipped for move to 1 model - spec_config_fixed = DraftTargetDecodingConfig( - max_draft_len=max_draft_len, - speculative_model=str(draft_model), - draft_len_schedule=None, # No schedule - fixed draft length + + if drafter_type == "model_drafter": + for batch_size, runtime_draft_len in runtime_schedule: + expected_draft_len = get_draft_len_for_batch_size(schedule, batch_size, max_draft_len) + assert runtime_draft_len == expected_draft_len, ( + f"DraftTarget resolved batch size {batch_size} to draft length " + f"{runtime_draft_len}, expected {expected_draft_len} from {schedule}" + ) + + runtime_transitions = [ + observation + for index, observation in enumerate(runtime_schedule) + if index == 0 or observation != runtime_schedule[index - 1] + ] + expected_key_transitions = {(8, 1), (4, 2), (1, 3)} + assert expected_key_transitions.issubset(runtime_transitions), ( + f"DraftTarget runtime schedule did not exercise {expected_key_transitions}: " + f"got {runtime_transitions}" ) + return + + # Reference: spec decode with fixed max_draft_len (no schedule) + spec_config_fixed = NGramDecodingConfig( + max_draft_len=max_draft_len, + max_matching_ngram_size=2, + draft_len_schedule=None, # No schedule - fixed draft length + is_keep_all=True, + is_use_oldest=True, + is_public_pool=False, + ) llm_fixed = LLM(**llm_common_config, speculative_config=spec_config_fixed) results_fixed = llm_fixed.generate(prompts, sampling_params_list) generated_text_fixed = [result.outputs[0].text for result in results_fixed] @@ -163,7 +226,7 @@ def test_draft_len_schedule_functionality( ) llm_common_config = dict( - model=llm_models_root() / "llama-3.1-model" / "Meta-Llama-3.1-8B", + model=llm_models_root() / "Qwen3" / "Qwen3-8B", backend="pytorch", attn_backend="TRTLLM", disable_overlap_scheduler=True, @@ -179,10 +242,9 @@ def test_draft_len_schedule_functionality( draft_len_schedule=draft_schedule, ) else: - # skipped for move to 1 model spec_config = DraftTargetDecodingConfig( max_draft_len=5, - speculative_model=str(llm_models_root() / "llama-3.2-models" / "Llama-3.2-3B-Instruct"), + speculative_model=str(llm_models_root() / "Qwen3" / "Qwen3-0.6B"), draft_len_schedule=draft_schedule, ) prompts = ["The capital of France is" for i in range(7)] diff --git a/tests/unittest/_torch/speculative/hw_agnostic/test_draft_target.py b/tests/unittest/_torch/speculative/hw_agnostic/test_draft_target.py index dd2b3e5cd7a0..721486a211c0 100644 --- a/tests/unittest/_torch/speculative/hw_agnostic/test_draft_target.py +++ b/tests/unittest/_torch/speculative/hw_agnostic/test_draft_target.py @@ -1,3 +1,18 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + import os import sys import unittest @@ -10,19 +25,18 @@ sys.path.append(os.path.join(os.path.dirname(__file__), "..")) from utils.llm_data import llm_models_root -from utils.util import similar @pytest.mark.parametrize("use_cuda_graph,attn_backend", [[False, "TRTLLM"], [True, "TRTLLM"]]) @pytest.mark.high_cuda_memory -def test_llama_draft_target(use_cuda_graph: bool, attn_backend: str): +def test_qwen3_draft_target(use_cuda_graph: bool, attn_backend: str): total_mem_gb = torch.cuda.get_device_properties(0).total_memory / 1e9 - if total_mem_gb < 60: - pytest.skip("Not enough memory to load target model") + if total_mem_gb < 30: + pytest.skip("Not enough memory to load target and draft models") models_path = llm_models_root() - draft_model_dir = f"{models_path}/llama-3.1-model/Llama-3.1-8B-Instruct" - target_model_dir = f"{models_path}/llama-3.1-model/Llama-3.1-8B-Instruct" + draft_model_dir = f"{models_path}/Qwen3/Qwen3-0.6B" + target_model_dir = f"{models_path}/Qwen3/Qwen3-8B" max_batch_size = 2 max_draft_len = 4 @@ -49,35 +63,42 @@ def test_llama_draft_target(use_cuda_graph: bool, attn_backend: str): "The capital of France is", "The president of the United States is", ] - sampling_params = SamplingParams(max_tokens=32, temperature=0.0) + # Eight tokens require at least two DraftTarget iterations while avoiding + # later shape-sensitive BF16 greedy ties observed on L40S. + max_tokens = 8 + sampling_params = SamplingParams(max_tokens=max_tokens, temperature=0.0) llm_spec = LLM(**llm_common_config, speculative_config=spec_config) results_spec = llm_spec.generate(prompts, sampling_params) - generated_text_spec = [result.outputs[0].text for result in results_spec] llm_spec.shutdown() llm_ref = LLM(**llm_common_config) results_ref = llm_ref.generate(prompts, sampling_params) - generated_text_ref = [result.outputs[0].text for result in results_ref] llm_ref.shutdown() - for text_spec, text_ref in zip(generated_text_spec, generated_text_ref): - # The spec decode algorithm currently guarantees identical results - assert similar(text_spec, text_ref) + for prompt, result_spec, result_ref in zip(prompts, results_spec, results_ref, strict=True): + spec_output = result_spec.outputs[0] + ref_output = result_ref.outputs[0] + assert len(spec_output.token_ids) == max_tokens + assert len(ref_output.token_ids) == max_tokens + assert spec_output.token_ids == ref_output.token_ids, ( + f"DraftTarget output tokens differ from greedy reference for prompt {prompt!r}: " + f"speculative={spec_output.token_ids}, reference={ref_output.token_ids}" + ) @pytest.mark.high_cuda_memory -def test_llama_draft_target_rejection(): +def test_qwen3_draft_target_rejection(): """DraftTarget one-model with rejection sampling on: the rejection path (draft-prob capture -> fail-closed guard -> rejection acceptance) runs end-to-end with non-greedy sampling and produces coherent output.""" total_mem_gb = torch.cuda.get_device_properties(0).total_memory / 1e9 - if total_mem_gb < 60: - pytest.skip("Not enough memory to load target model") + if total_mem_gb < 30: + pytest.skip("Not enough memory to load target and draft models") models_path = llm_models_root() - target_model_dir = f"{models_path}/llama-3.1-model/Llama-3.1-8B-Instruct" - draft_model_dir = f"{models_path}/llama-3.2-models/Llama-3.2-1B-Instruct" + target_model_dir = f"{models_path}/Qwen3/Qwen3-8B" + draft_model_dir = f"{models_path}/Qwen3/Qwen3-0.6B" spec_config = DraftTargetDecodingConfig( max_draft_len=4, diff --git a/tests/unittest/_torch/speculative/hw_agnostic/test_save_state.py b/tests/unittest/_torch/speculative/hw_agnostic/test_save_state.py index 26663704dce6..afff1c36b87b 100644 --- a/tests/unittest/_torch/speculative/hw_agnostic/test_save_state.py +++ b/tests/unittest/_torch/speculative/hw_agnostic/test_save_state.py @@ -1,3 +1,18 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + import os import sys import tempfile @@ -12,6 +27,8 @@ sys.path.append(os.path.join(os.path.dirname(__file__), "..")) +QWEN3_5_HIDDEN_SIZE = 2560 + def test_multi_save_state(): use_cuda_graph = True @@ -22,12 +39,12 @@ def test_multi_save_state(): layers_to_capture = {10, 11, 12} total_mem_gb = torch.cuda.get_device_properties(0).total_memory / 1e9 - if total_mem_gb < 80: - pytest.skip("Not enough memory to load target + draft model") + if total_mem_gb < 20: + pytest.skip("Not enough memory to load target model") models_path = llm_models_root() with tempfile.TemporaryDirectory() as temp_dir: - target_model_dir = f"{models_path}/llama-3.2-models/Llama-3.2-1B-Instruct" + target_model_dir = f"{models_path}/Qwen3.5-4B" max_batch_size = 16 kv_cache_config = KvCacheConfig( @@ -65,9 +82,9 @@ def test_multi_save_state(): assert saved_data["aux_hidden_states"].shape == ( len(tok_ids), - 2048 * len(layers_to_capture), + QWEN3_5_HIDDEN_SIZE * len(layers_to_capture), ) - assert saved_data["hidden_state"].shape == (len(tok_ids), 2048) + assert saved_data["hidden_state"].shape == (len(tok_ids), QWEN3_5_HIDDEN_SIZE) assert saved_data["input_ids"].tolist() == tok_ids @@ -80,12 +97,12 @@ def test_save_state(layers_to_capture): enable_chunked_prefill = False total_mem_gb = torch.cuda.get_device_properties(0).total_memory / 1e9 - if total_mem_gb < 80: - pytest.skip("Not enough memory to load target + draft model") + if total_mem_gb < 20: + pytest.skip("Not enough memory to load target model") models_path = llm_models_root() with tempfile.TemporaryDirectory() as temp_dir: - target_model_dir = f"{models_path}/llama-3.2-models/Llama-3.2-1B-Instruct" + target_model_dir = f"{models_path}/Qwen3.5-4B" max_batch_size = 16 kv_cache_config = KvCacheConfig( @@ -121,12 +138,21 @@ def test_save_state(layers_to_capture): # Read in .pt file saved_data = torch.load(os.path.join(temp_dir, "data_1.pt"))[0] if layers_to_capture is None: - assert saved_data["aux_hidden_states"].shape == (len(tok_ids), 2048 * 3) - assert saved_data["hidden_state"].shape == (len(tok_ids), 2048) + assert saved_data["aux_hidden_states"].shape == ( + len(tok_ids), + QWEN3_5_HIDDEN_SIZE * 3, + ) + assert saved_data["hidden_state"].shape == ( + len(tok_ids), + QWEN3_5_HIDDEN_SIZE, + ) assert saved_data["input_ids"].tolist() == tok_ids else: assert "aux_hidden_states" not in saved_data - assert saved_data["hidden_state"].shape == (len(tok_ids), 2048) + assert saved_data["hidden_state"].shape == ( + len(tok_ids), + QWEN3_5_HIDDEN_SIZE, + ) assert saved_data["input_ids"].tolist() == tok_ids diff --git a/tests/unittest/auto_deploy/_utils_test/_model_test_utils.py b/tests/unittest/auto_deploy/_utils_test/_model_test_utils.py index 62253b7069e7..9b726263e174 100644 --- a/tests/unittest/auto_deploy/_utils_test/_model_test_utils.py +++ b/tests/unittest/auto_deploy/_utils_test/_model_test_utils.py @@ -447,24 +447,6 @@ def apply_rotary_pos_emb_ds(q, k, cos, sin, position_ids, unsqueeze_dim=1): "num_experts": 16, }, }, - "meta-llama/Llama-4-Scout-17B-16E-Instruct": { - "model_factory": "AutoModelForImageTextToText", - "model_kwargs": { - "text_config": { - "num_hidden_layers": 1, - "head_dim": 64, - "hidden_size": 32, - "intermediate_size": 64, - "intermediate_size_mlp": 64, - "num_attention_heads": 2, - "num_key_value_heads": 1, - "num_local_experts": 2, - }, - "vision_config": { - "num_hidden_layers": 1, - }, - }, - }, "deepseek-ai/DeepSeek-V3": { "model_kwargs": { "first_k_dense_replace": 1, diff --git a/tests/unittest/auto_deploy/singlegpu/models/test_decilm_modeling.py b/tests/unittest/auto_deploy/singlegpu/models/test_decilm_modeling.py deleted file mode 100644 index 6d23313ad070..000000000000 --- a/tests/unittest/auto_deploy/singlegpu/models/test_decilm_modeling.py +++ /dev/null @@ -1,558 +0,0 @@ -# SPDX-FileCopyrightText: Copyright (c) 2022-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -# SPDX-License-Identifier: Apache-2.0 -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -"""Tests for DeciLM (Nemotron-NAS) custom model implementation. - -Tests the custom AD model against an inline HF reference implementation. -The HF config is loaded via trust_remote_code=True at runtime; for unit tests -we create a small config using the same class loaded from the HF checkpoint. -""" - -import math - -import pytest -import torch -import torch.nn as nn -import torch.nn.functional as F -from _model_test_utils import assert_rmse_close -from torch.export import Dim -from transformers import AutoConfig - -from tensorrt_llm._torch.auto_deploy.export import torch_export_to_gm -from tensorrt_llm._torch.auto_deploy.models.custom.modeling_decilm import ( - DeciLMAttention, - DeciLMDecoderLayer, - DeciLMForCausalLM, - DeciLMMLP, - DeciLMRotaryEmbedding, - _ffn_mult_to_intermediate_size, -) -from tensorrt_llm._torch.auto_deploy.utils._graph import move_to_device - -_BATCH_AND_SEQUENCE_TEST_CASES = ((2, 6), (1, 8)) - - -@pytest.fixture(scope="function", autouse=True) -def set_seed(): - torch.manual_seed(42) - - -# ============================================================================= -# Small test config (mimics HF DeciLMConfig structure loaded via trust_remote_code) -# ============================================================================= - -# Block configs for a small test model: -# Layer 0: attention (GQA, n_heads_in_group=2) + FFN (ffn_mult=2.625) -# Layer 1: FFN-only (attention no_op, ffn_mult=1.3125) -# Layer 2: attention (GQA, n_heads_in_group=2) + FFN (ffn_mult=5.25) -_SMALL_BLOCK_CONFIGS = [ - { - "attention": {"n_heads_in_group": 2, "no_op": False}, - "ffn": {"ffn_mult": 2.625, "no_op": False}, - }, - { - "attention": {"no_op": True}, - "ffn": {"ffn_mult": 1.3125, "no_op": False}, - }, - { - "attention": {"n_heads_in_group": 2, "no_op": False}, - "ffn": {"ffn_mult": 5.25, "no_op": False}, - }, -] - - -def _create_small_config(): - """Create a small DeciLM config for testing. - - Uses AutoConfig with trust_remote_code=True to load the HF DeciLMConfig class, - then creates a small instance for testing. - """ - try: - hf_config = AutoConfig.from_pretrained( - "nvidia/Llama-3_3-Nemotron-Super-49B-v1", trust_remote_code=True - ) - ConfigCls = type(hf_config) - except Exception: - pytest.skip("Cannot load DeciLMConfig from HF (network or cache unavailable)") - - config = ConfigCls( - vocab_size=1000, - hidden_size=64, - num_hidden_layers=3, - num_attention_heads=4, - hidden_act="silu", - max_position_embeddings=512, - rms_norm_eps=1e-5, - attention_bias=False, - mlp_bias=False, - rope_theta=500000.0, - rope_scaling={ - "rope_type": "llama3", - "factor": 8.0, - "high_freq_factor": 4.0, - "low_freq_factor": 1.0, - "original_max_position_embeddings": 64, - }, - block_configs=_SMALL_BLOCK_CONFIGS, - ) - config._attn_implementation = "eager" - return config - - -# ============================================================================= -# HF Reference Classes (standalone, for equivalence testing) -# ============================================================================= - - -class _HFDeciLMRMSNorm(nn.Module): - def __init__(self, hidden_size, eps=1e-6): - super().__init__() - self.weight = nn.Parameter(torch.ones(hidden_size)) - self.variance_epsilon = eps - - def forward(self, hidden_states): - input_dtype = hidden_states.dtype - hidden_states = hidden_states.to(torch.float32) - variance = hidden_states.pow(2).mean(-1, keepdim=True) - hidden_states = hidden_states * torch.rsqrt(variance + self.variance_epsilon) - return self.weight * hidden_states.to(input_dtype) - - -class _HFDeciLMRotaryEmbedding(nn.Module): - def __init__(self, config): - super().__init__() - head_dim = config.hidden_size // config.num_attention_heads - base = config.rope_theta - inv_freq = 1.0 / (base ** (torch.arange(0, head_dim, 2, dtype=torch.float32) / head_dim)) - - if config.rope_scaling is not None: - rope_type = config.rope_scaling.get("rope_type", "default") - if rope_type == "llama3": - factor = config.rope_scaling["factor"] - low_freq_factor = config.rope_scaling.get("low_freq_factor", 1.0) - high_freq_factor = config.rope_scaling.get("high_freq_factor", 4.0) - old_context_len = config.rope_scaling.get("original_max_position_embeddings", 8192) - low_freq_wavelen = old_context_len / low_freq_factor - high_freq_wavelen = old_context_len / high_freq_factor - wavelen = 2 * math.pi / inv_freq - inv_freq_scaled = inv_freq / factor - smooth_factor = (old_context_len / wavelen - low_freq_factor) / ( - high_freq_factor - low_freq_factor - ) - smooth_factor = torch.clamp(smooth_factor, 0.0, 1.0) - smoothed = (1 - smooth_factor) * inv_freq_scaled + smooth_factor * inv_freq - inv_freq = torch.where(wavelen < high_freq_wavelen, inv_freq, smoothed) - inv_freq = torch.where(wavelen > low_freq_wavelen, inv_freq_scaled, inv_freq) - - self.register_buffer("inv_freq", inv_freq, persistent=False) - - @torch.no_grad() - def forward(self, x, position_ids): - inv_freq_expanded = ( - self.inv_freq[None, :, None].float().expand(position_ids.shape[0], -1, 1) - ) - position_ids_expanded = position_ids[:, None, :].float() - freqs = (inv_freq_expanded.float() @ position_ids_expanded.float()).transpose(1, 2) - emb = torch.cat((freqs, freqs), dim=-1) - return emb.cos().to(dtype=x.dtype), emb.sin().to(dtype=x.dtype) - - -def _hf_rotate_half(x): - x1 = x[..., : x.shape[-1] // 2] - x2 = x[..., x.shape[-1] // 2 :] - return torch.cat((-x2, x1), dim=-1) - - -def _hf_apply_rotary_pos_emb(q, k, cos, sin, unsqueeze_dim=1): - cos = cos.unsqueeze(unsqueeze_dim) - sin = sin.unsqueeze(unsqueeze_dim) - q_embed = (q * cos) + (_hf_rotate_half(q) * sin) - k_embed = (k * cos) + (_hf_rotate_half(k) * sin) - return q_embed, k_embed - - -def _hf_repeat_kv(hidden_states, n_rep): - batch, num_kv_heads, slen, head_dim = hidden_states.shape - if n_rep == 1: - return hidden_states - hidden_states = hidden_states[:, :, None, :, :].expand( - batch, num_kv_heads, n_rep, slen, head_dim - ) - return hidden_states.reshape(batch, num_kv_heads * n_rep, slen, head_dim) - - -class _HFDeciLMAttention(nn.Module): - def __init__(self, config, n_heads_in_group, layer_idx): - super().__init__() - self.hidden_size = config.hidden_size - self.num_heads = config.num_attention_heads - self.head_dim = self.hidden_size // self.num_heads - self.num_key_value_groups = n_heads_in_group - self.num_key_value_heads = self.num_heads // self.num_key_value_groups - - self.q_proj = nn.Linear(self.hidden_size, self.num_heads * self.head_dim, bias=False) - self.k_proj = nn.Linear( - self.hidden_size, self.num_key_value_heads * self.head_dim, bias=False - ) - self.v_proj = nn.Linear( - self.hidden_size, self.num_key_value_heads * self.head_dim, bias=False - ) - self.o_proj = nn.Linear(self.hidden_size, self.hidden_size, bias=False) - self.rotary_emb = _HFDeciLMRotaryEmbedding(config) - - def forward(self, hidden_states, position_ids): - bsz, q_len, _ = hidden_states.size() - query_states = ( - self.q_proj(hidden_states) - .view(bsz, q_len, self.num_heads, self.head_dim) - .transpose(1, 2) - ) - key_states = ( - self.k_proj(hidden_states) - .view(bsz, q_len, self.num_key_value_heads, self.head_dim) - .transpose(1, 2) - ) - value_states = ( - self.v_proj(hidden_states) - .view(bsz, q_len, self.num_key_value_heads, self.head_dim) - .transpose(1, 2) - ) - - cos, sin = self.rotary_emb(value_states, position_ids) - query_states, key_states = _hf_apply_rotary_pos_emb(query_states, key_states, cos, sin) - key_states = _hf_repeat_kv(key_states, self.num_key_value_groups) - value_states = _hf_repeat_kv(value_states, self.num_key_value_groups) - - attn_weights = torch.matmul(query_states, key_states.transpose(2, 3)) / math.sqrt( - self.head_dim - ) - causal_mask = torch.triu( - torch.full( - (q_len, q_len), - float("-inf"), - device=hidden_states.device, - dtype=hidden_states.dtype, - ), - diagonal=1, - ) - attn_weights = attn_weights + causal_mask - attn_weights = F.softmax(attn_weights, dim=-1, dtype=torch.float32).to(query_states.dtype) - attn_output = torch.matmul(attn_weights, value_states) - attn_output = attn_output.transpose(1, 2).contiguous().reshape(bsz, q_len, -1) - return self.o_proj(attn_output) - - -class _HFDeciLMMLP(nn.Module): - def __init__(self, config, ffn_mult): - super().__init__() - intermediate_size = _ffn_mult_to_intermediate_size(ffn_mult, config.hidden_size) - self.gate_proj = nn.Linear(config.hidden_size, intermediate_size, bias=config.mlp_bias) - self.up_proj = nn.Linear(config.hidden_size, intermediate_size, bias=config.mlp_bias) - self.down_proj = nn.Linear(intermediate_size, config.hidden_size, bias=config.mlp_bias) - - def forward(self, x): - return self.down_proj(F.silu(self.gate_proj(x)) * self.up_proj(x)) - - -class _HFDeciLMDecoderLayer(nn.Module): - def __init__(self, config, layer_idx): - super().__init__() - block_config = config.block_configs[layer_idx] - self.has_attention = not block_config.attention.no_op - self.has_ffn = not block_config.ffn.no_op - - if self.has_attention: - self.input_layernorm = _HFDeciLMRMSNorm(config.hidden_size, eps=config.rms_norm_eps) - self.self_attn = _HFDeciLMAttention( - config, block_config.attention.n_heads_in_group, layer_idx - ) - if self.has_ffn: - self.post_attention_layernorm = _HFDeciLMRMSNorm( - config.hidden_size, eps=config.rms_norm_eps - ) - self.mlp = _HFDeciLMMLP(config, block_config.ffn.ffn_mult) - - def forward(self, hidden_states, position_ids): - if self.has_attention: - residual = hidden_states - hidden_states = self.input_layernorm(hidden_states) - hidden_states = self.self_attn(hidden_states, position_ids) - hidden_states = residual + hidden_states - if self.has_ffn: - residual = hidden_states - hidden_states = self.post_attention_layernorm(hidden_states) - hidden_states = self.mlp(hidden_states) - hidden_states = residual + hidden_states - return hidden_states - - -class _HFDeciLMForCausalLM(nn.Module): - def __init__(self, config): - super().__init__() - self.embed_tokens = nn.Embedding(config.vocab_size, config.hidden_size) - self.layers = nn.ModuleList( - [_HFDeciLMDecoderLayer(config, idx) for idx in range(config.num_hidden_layers)] - ) - self.norm = _HFDeciLMRMSNorm(config.hidden_size, eps=config.rms_norm_eps) - self.lm_head = nn.Linear(config.hidden_size, config.vocab_size, bias=False) - - def forward(self, input_ids, position_ids): - hidden_states = self.embed_tokens(input_ids) - for layer in self.layers: - hidden_states = layer(hidden_states, position_ids) - hidden_states = self.norm(hidden_states) - return self.lm_head(hidden_states).float() - - -# ============================================================================= -# Unit Tests: Structure -# ============================================================================= - - -@pytest.mark.cpu_only -def test_decilm_layer_structure(): - """Test that layers have correct structure based on block_configs.""" - config = _create_small_config() - model = DeciLMForCausalLM(config) - - layer0 = model.model.layers[0] - assert layer0.has_attention is True - assert hasattr(layer0, "self_attn") - assert hasattr(layer0, "input_layernorm") - - layer1 = model.model.layers[1] - assert layer1.has_attention is False - assert not hasattr(layer1, "self_attn") - - layer2 = model.model.layers[2] - assert layer2.has_attention is True - - -@pytest.mark.cpu_only -def test_decilm_ffn_mult_to_intermediate_size(): - assert _ffn_mult_to_intermediate_size(5.25, 8192) == 28672 - assert _ffn_mult_to_intermediate_size(2.625, 8192) == 14336 - assert _ffn_mult_to_intermediate_size(1.0, 8192) == 5632 - assert _ffn_mult_to_intermediate_size(0.5, 8192) == 2816 - - -@pytest.mark.cpu_only -def test_decilm_weight_keys_match_checkpoint(): - config = _create_small_config() - model = DeciLMForCausalLM(config) - keys = set(model.state_dict().keys()) - - assert "model.layers.0.self_attn.q_proj.weight" in keys - assert "model.layers.0.input_layernorm.weight" in keys - assert "model.layers.1.self_attn.q_proj.weight" not in keys - assert "model.layers.1.input_layernorm.weight" not in keys - assert "model.layers.1.post_attention_layernorm.weight" in keys - assert "model.layers.1.mlp.gate_proj.weight" in keys - - -# ============================================================================= -# Numerical Equivalence Tests -# ============================================================================= - - -@pytest.mark.parametrize("B,S", _BATCH_AND_SEQUENCE_TEST_CASES) -@pytest.mark.parametrize("dtype", [torch.bfloat16]) -@torch.no_grad() -def test_decilm_mlp_equivalence(B, S, dtype): - device = "cuda" - config = _create_small_config() - ffn_mult = 2.625 - intermediate_size = _ffn_mult_to_intermediate_size(ffn_mult, config.hidden_size) - - hf_mlp = _HFDeciLMMLP(config, ffn_mult).to(device=device, dtype=dtype).eval() - custom_mlp = DeciLMMLP(config, intermediate_size).to(device=device, dtype=dtype) - custom_mlp.load_state_dict(hf_mlp.state_dict()) - custom_mlp.eval() - - x = torch.randn(B, S, config.hidden_size, device=device, dtype=dtype) - torch.testing.assert_close(custom_mlp(x), hf_mlp(x), rtol=1e-3, atol=1e-3) - - -@pytest.mark.parametrize("B,S", _BATCH_AND_SEQUENCE_TEST_CASES) -@pytest.mark.parametrize("dtype", [torch.bfloat16]) -@torch.no_grad() -def test_decilm_attention_equivalence(B, S, dtype): - device = "cuda" - config = _create_small_config() - n_heads_in_group = 2 - - hf_attn = _HFDeciLMAttention(config, n_heads_in_group, 0).to(device=device, dtype=dtype).eval() - custom_attn = DeciLMAttention(config, n_heads_in_group, 0).to(device=device, dtype=dtype) - custom_sd = {k: v for k, v in hf_attn.state_dict().items() if not k.startswith("rotary_emb.")} - custom_attn.load_state_dict(custom_sd) - custom_attn.eval() - - x = torch.randn(B, S, config.hidden_size, device=device, dtype=dtype) - position_ids = torch.arange(S, device=device).unsqueeze(0).expand(B, -1) - - hf_out = hf_attn(x, position_ids) - - rotary_emb = DeciLMRotaryEmbedding(config).to(device=device, dtype=dtype) - position_embeddings = rotary_emb(x, position_ids) - custom_out = custom_attn(x, position_embeddings) - - assert_rmse_close(custom_out, hf_out, rmse_ratio_tol=0.10, msg="Attention: ") - - -@pytest.mark.parametrize("B,S", _BATCH_AND_SEQUENCE_TEST_CASES) -@pytest.mark.parametrize("dtype", [torch.bfloat16]) -@torch.no_grad() -def test_decilm_decoder_layer_equivalence(B, S, dtype): - device = "cuda" - config = _create_small_config() - layer_idx = 0 - - hf_layer = _HFDeciLMDecoderLayer(config, layer_idx).to(device=device, dtype=dtype).eval() - custom_layer = DeciLMDecoderLayer(config, layer_idx).to(device=device, dtype=dtype) - custom_sd = {k: v for k, v in hf_layer.state_dict().items() if "rotary_emb." not in k} - custom_layer.load_state_dict(custom_sd) - custom_layer.eval() - - x = torch.randn(B, S, config.hidden_size, device=device, dtype=dtype) - position_ids = torch.arange(S, device=device).unsqueeze(0).expand(B, -1) - - hf_out = hf_layer(x, position_ids) - - rotary_emb = DeciLMRotaryEmbedding(config).to(device=device, dtype=dtype) - position_embeddings = rotary_emb(x, position_ids) - custom_out = custom_layer(x, position_embeddings) - - assert_rmse_close(custom_out, hf_out, rmse_ratio_tol=0.05, msg="Decoder layer: ") - - -@pytest.mark.parametrize("B,S", _BATCH_AND_SEQUENCE_TEST_CASES) -@pytest.mark.parametrize("dtype", [torch.bfloat16]) -@torch.no_grad() -def test_decilm_ffn_only_layer_equivalence(B, S, dtype): - device = "cuda" - config = _create_small_config() - layer_idx = 1 # FFN-only - - hf_layer = _HFDeciLMDecoderLayer(config, layer_idx).to(device=device, dtype=dtype).eval() - custom_layer = DeciLMDecoderLayer(config, layer_idx).to(device=device, dtype=dtype) - custom_layer.load_state_dict(hf_layer.state_dict()) - custom_layer.eval() - - x = torch.randn(B, S, config.hidden_size, device=device, dtype=dtype) - position_ids = torch.arange(S, device=device).unsqueeze(0).expand(B, -1) - - hf_out = hf_layer(x, position_ids) - - rotary_emb = DeciLMRotaryEmbedding(config).to(device=device, dtype=dtype) - position_embeddings = rotary_emb(x, position_ids) - custom_out = custom_layer(x, position_embeddings) - - torch.testing.assert_close(custom_out, hf_out, rtol=1e-3, atol=1e-3) - - -@pytest.mark.parametrize("B,S", _BATCH_AND_SEQUENCE_TEST_CASES) -@pytest.mark.parametrize("dtype", [torch.bfloat16]) -@torch.no_grad() -def test_decilm_full_model_equivalence(B, S, dtype): - device = "cuda" - config = _create_small_config() - - hf_model = _HFDeciLMForCausalLM(config).to(device=device, dtype=dtype).eval() - custom_model = DeciLMForCausalLM(config).to(device=device, dtype=dtype) - - hf_sd = hf_model.state_dict() - # Map HF reference keys to custom model keys (model. prefix for custom) - custom_sd = {} - for k, v in hf_sd.items(): - if "rotary_emb." in k: - continue - if k.startswith("embed_tokens.") or k.startswith("layers.") or k.startswith("norm."): - custom_sd[f"model.{k}"] = v - else: - custom_sd[k] = v - - custom_expected = {k for k in custom_model.state_dict().keys() if "rotary_emb." not in k} - assert set(custom_sd.keys()) == custom_expected, ( - f"Key mismatch.\n" - f" Extra: {set(custom_sd.keys()) - custom_expected}\n" - f" Missing: {custom_expected - set(custom_sd.keys())}" - ) - custom_model.load_state_dict(custom_sd, strict=False) - custom_model.eval() - - input_ids = torch.randint(0, config.vocab_size, (B, S), device=device) - position_ids = torch.arange(S, device=device).unsqueeze(0).expand(B, -1) - - hf_logits = hf_model(input_ids, position_ids) - custom_out = custom_model(input_ids=input_ids, position_ids=position_ids) - - assert_rmse_close(custom_out.logits, hf_logits, rmse_ratio_tol=0.05, msg="Full model: ") - - -# ============================================================================= -# Export Test -# ============================================================================= - - -def test_decilm_model_can_be_exported(): - """Test export with torch_export_to_gm, dynamic shapes, and equivalence.""" - device = "cuda" - dtype = torch.bfloat16 - config = _create_small_config() - - model = DeciLMForCausalLM(config).to(device=device, dtype=dtype).eval() - - B, S = 2, 8 - input_ids = torch.randint(0, config.vocab_size, (B, S), device=device) - position_ids = torch.arange(S, device=device).unsqueeze(0).expand(B, -1) - - with torch.inference_mode(): - ref_out = model(input_ids=input_ids, position_ids=position_ids) - - dynamic_shapes = ( - {0: Dim.DYNAMIC, 1: Dim.DYNAMIC}, - {0: Dim.DYNAMIC, 1: Dim.DYNAMIC}, - ) - - gm = torch_export_to_gm( - model, - args=tuple(), - kwargs={"input_ids": input_ids, "position_ids": position_ids}, - dynamic_shapes=dynamic_shapes, - ) - move_to_device(gm, device) - - with torch.inference_mode(): - out_gm = gm(input_ids=input_ids, position_ids=position_ids) - - assert "logits" in out_gm - logits = out_gm["logits"] - assert logits.shape == (B, S, config.vocab_size) - assert torch.isfinite(logits).all() - assert_rmse_close(logits, ref_out.logits, rmse_ratio_tol=0.05, msg="Export: ") - - # Test different shape - B2, S2 = 1, 4 - input_ids2 = torch.randint(0, config.vocab_size, (B2, S2), device=device) - position_ids2 = torch.arange(S2, device=device).unsqueeze(0).expand(B2, -1) - - with torch.inference_mode(): - ref_out2 = model(input_ids=input_ids2, position_ids=position_ids2) - out_gm2 = gm(input_ids=input_ids2, position_ids=position_ids2) - - logits2 = out_gm2["logits"] - assert logits2.shape == (B2, S2, config.vocab_size) - assert torch.isfinite(logits2).all() - assert_rmse_close(logits2, ref_out2.logits, rmse_ratio_tol=0.05, msg="Export dynamic: ") diff --git a/tests/unittest/auto_deploy/singlegpu/smoke/test_ad_build_small_single.py b/tests/unittest/auto_deploy/singlegpu/smoke/test_ad_build_small_single.py index 993614dcbe00..67ca8b41dfd1 100644 --- a/tests/unittest/auto_deploy/singlegpu/smoke/test_ad_build_small_single.py +++ b/tests/unittest/auto_deploy/singlegpu/smoke/test_ad_build_small_single.py @@ -126,27 +126,6 @@ def _check_ad_config(experiment_config: ExperimentConfig, llm_args: LlmArgs): "mode": "transformers", }, ), - ( - "meta-llama/Llama-4-Scout-17B-16E-Instruct", - { - "transforms": { - "insert_cached_attention": {"backend": "flashinfer"}, - "compile_model": { - "backend": "torch-simple", - "piecewise_enabled": False, - }, - }, - }, - ), - ( - "meta-llama/Llama-4-Scout-17B-16E-Instruct", - { - "transforms": { - "transformers_replace_cached_attn": {"backend": "flashinfer"}, - }, - "mode": "transformers", - }, - ), ( "deepseek-ai/DeepSeek-V3", { diff --git a/tests/unittest/llmapi/apps/_test_openai_lora.py b/tests/unittest/llmapi/apps/_test_openai_lora.py index f4baa0ccb69e..36992cfcbc08 100644 --- a/tests/unittest/llmapi/apps/_test_openai_lora.py +++ b/tests/unittest/llmapi/apps/_test_openai_lora.py @@ -1,7 +1,23 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + import os import tempfile from dataclasses import asdict -from typing import List, Optional +from pathlib import Path +from typing import Generator import openai import pytest @@ -9,23 +25,22 @@ from tensorrt_llm.executor.request import LoRARequest +from ..lora_test_utils import qwen3_lora_adapter from ..test_llm import get_model_path from .openai_server import RemoteOpenAIServer pytestmark = pytest.mark.threadleak(enabled=False) -@pytest.fixture(scope="module", ids=["llama-models/llama-7b-hf"]) +@pytest.fixture(scope="module", ids=["Qwen3/Qwen3-0.6B"]) def model_name() -> str: - return "llama-models/llama-7b-hf" + return "Qwen3/Qwen3-0.6B" @pytest.fixture(scope="module") -def lora_adapter_names() -> List[Optional[str]]: - return [ - None, "llama-models/luotuo-lora-7b-0.1", - "llama-models/Japanese-Alpaca-LoRA-7b-v0" - ] +def lora_adapter_path() -> Generator[Path, None, None]: + with qwen3_lora_adapter() as adapter_path: + yield adapter_path @pytest.fixture(scope="module") @@ -71,43 +86,46 @@ def client(server: RemoteOpenAIServer) -> openai.OpenAI: def test_lora(client: openai.OpenAI, model_name: str, - lora_adapter_names: List[str]): - prompts = [ - "美国的首都在哪里? \n答案:", - "美国的首都在哪里? \n答案:", - "美国的首都在哪里? \n答案:", - "アメリカ合衆国の首都はどこですか? \n答え:", - "アメリカ合衆国の首都はどこですか? \n答え:", - "アメリカ合衆国の首都はどこですか? \n答え:", - ] - references = [ - "沃尔玛\n\n## 新闻\n\n* ", - "美国的首都是华盛顿。\n\n美国的", - "纽约\n\n### カンファレンスの", - "Washington, D.C.\nWashington, D.C. is the capital of the United", - "华盛顿。\n\n英国の首都是什", - "ワシントン\nQ1. アメリカ合衆国", - ] - - for prompt, reference, lora_adapter_name in zip(prompts, references, - lora_adapter_names * 2): - extra_body = {} - if lora_adapter_name is not None: - lora_req = LoRARequest(lora_adapter_name, - lora_adapter_names.index(lora_adapter_name), - get_model_path(lora_adapter_name)) - extra_body["lora_request"] = asdict(lora_req) + lora_adapter_path: Path) -> None: + prompt = "The capital of France is" + def complete( + extra_body: dict[str, object] | None = None + ) -> tuple[str, tuple[str, ...], tuple[float | None, ...]]: response = client.completions.create( model=model_name, prompt=prompt, max_tokens=20, temperature=0.0, - extra_body=extra_body, + logprobs=1, + extra_body=extra_body or {}, + ) + choice = response.choices[0] + assert choice.logprobs is not None + assert choice.logprobs.tokens is not None + assert choice.logprobs.token_logprobs is not None + return ( + choice.text, + tuple(choice.logprobs.tokens), + tuple(choice.logprobs.token_logprobs), ) - output = response.choices[0].text - print(f"response: {output}") - print(f"reference: {reference}") - assert output == reference, ( - f"Unexpected output for LoRA adapter {lora_adapter_name!r}: " - f"prompt={prompt!r}, response={output!r}, reference={reference!r}") + + base_output = complete() + lora_request = LoRARequest(lora_name=lora_adapter_path.name, + lora_int_id=1, + lora_path=str(lora_adapter_path)) + extra_body = {"lora_request": asdict(lora_request)} + first_lora_output = complete(extra_body) + reused_lora_output = complete(extra_body) + + assert base_output[0] + assert first_lora_output[0] + output_changed = first_lora_output[:2] != base_output[:2] + logprobs_changed = first_lora_output[2] != pytest.approx(base_output[2], + abs=1e-4) + assert output_changed or logprobs_changed + assert reused_lora_output[:2] == first_lora_output[:2] + # Adapter reuse can shift same-token logprobs by about 0.06 due to + # numerically equivalent kernel execution, but larger drift is suspicious. + assert reused_lora_output[2] == pytest.approx(first_lora_output[2], + abs=0.075) diff --git a/tests/unittest/llmapi/apps/_test_openai_multi_gpu.py b/tests/unittest/llmapi/apps/_test_openai_multi_gpu.py index f2f06550c52b..56b9b2218c64 100644 --- a/tests/unittest/llmapi/apps/_test_openai_multi_gpu.py +++ b/tests/unittest/llmapi/apps/_test_openai_multi_gpu.py @@ -1,3 +1,18 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + import os import tempfile @@ -12,7 +27,7 @@ @pytest.fixture(scope="module") def model_name(): - return "llama-models-v3/llama-v3-8b-instruct-hf" + return "Qwen3.5-4B" @pytest.fixture(scope="module", params=["pytorch"]) @@ -90,7 +105,7 @@ def test_chat_tp2(client: openai.OpenAI, model_name: str): assert len(chat_completion.choices) == 1 assert chat_completion.usage.completion_tokens == 1 message = chat_completion.choices[0].message - assert message.content == "Two" + assert message is not None @skip_single_gpu @@ -102,7 +117,8 @@ def test_completion_tp2(client: openai.OpenAI, model_name: str): max_tokens=5, temperature=0.0, ) - assert completion.choices[0].text == " D E F G H" + assert completion.choices + assert completion.usage.completion_tokens == 5 @skip_single_gpu @@ -127,8 +143,6 @@ async def test_chat_streaming_tp2(async_client: openai.AsyncOpenAI, delta = chunk.choices[0].delta if delta.role: assert delta.role == "assistant" - if delta.content: - assert delta.content == "Two" @skip_single_gpu diff --git a/tests/unittest/llmapi/apps/_test_trtllm_serve_lora.py b/tests/unittest/llmapi/apps/_test_trtllm_serve_lora.py index e94c30662b1a..29f290e2f752 100644 --- a/tests/unittest/llmapi/apps/_test_trtllm_serve_lora.py +++ b/tests/unittest/llmapi/apps/_test_trtllm_serve_lora.py @@ -1,24 +1,49 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + import os -import subprocess -import sys import tempfile +from dataclasses import asdict +from pathlib import Path +from typing import Generator +import openai import pytest import yaml -from .openai_server import RemoteOpenAIServer +from tensorrt_llm.executor.request import LoRARequest -sys.path.append(os.path.join(os.path.dirname(__file__), '..')) -from test_llm import get_model_path +from ..lora_test_utils import qwen3_lora_adapter +from ..test_llm import get_model_path +from .openai_server import RemoteOpenAIServer -@pytest.fixture(scope="module", ids=["llama-models/llama-7b-hf"]) +@pytest.fixture(scope="module", ids=["Qwen3/Qwen3-0.6B"]) def model_name() -> str: - return "llama-models/llama-7b-hf" + return "Qwen3/Qwen3-0.6B" @pytest.fixture(scope="module") -def temp_extra_llm_api_options_file(): +def lora_adapter_path() -> Generator[Path, None, None]: + with qwen3_lora_adapter() as adapter_path: + yield adapter_path + + +@pytest.fixture(scope="module") +def temp_extra_llm_api_options_file( + lora_adapter_path: Path) -> Generator[str, None, None]: temp_dir = tempfile.gettempdir() temp_file_path = os.path.join(temp_dir, "extra_llm_api_options.yaml") try: @@ -41,7 +66,9 @@ def temp_extra_llm_api_options_file(): @pytest.fixture(scope="module") -def server(model_name: str, temp_extra_llm_api_options_file: str): +def server( + model_name: str, temp_extra_llm_api_options_file: str +) -> Generator[RemoteOpenAIServer, None, None]: model_path = get_model_path(model_name) args = [ "--backend", "pytorch", "--extra_llm_api_options", @@ -53,19 +80,20 @@ def server(model_name: str, temp_extra_llm_api_options_file: str): @pytest.fixture(scope="module") -def example_root(): - llm_root = os.getenv("LLM_ROOT") - return os.path.join(llm_root, "examples", "serve") - - -@pytest.mark.parametrize("exe, script", - [("python3", "openai_completion_client_for_lora.py")]) -def test_trtllm_serve_examples(exe: str, script: str, - server: RemoteOpenAIServer, example_root: str): - client_script = os.path.join(example_root, script) - # CalledProcessError will be raised if any errors occur - subprocess.run([exe, client_script], - stdout=subprocess.PIPE, - stderr=subprocess.PIPE, - text=True, - check=True) +def client(server: RemoteOpenAIServer) -> openai.OpenAI: + return server.get_client() + + +def test_trtllm_serve_lora_completion(client: openai.OpenAI, model_name: str, + lora_adapter_path: Path) -> None: + lora_request = LoRARequest(lora_name=lora_adapter_path.name, + lora_int_id=0, + lora_path=str(lora_adapter_path)) + response = client.completions.create( + model=model_name, + prompt="The capital of France is", + max_tokens=20, + extra_body={"lora_request": asdict(lora_request)}, + ) + + assert response.choices[0].text diff --git a/tests/unittest/llmapi/lora_test_utils.py b/tests/unittest/llmapi/lora_test_utils.py index da6b20451fc6..d2e63a8a9326 100644 --- a/tests/unittest/llmapi/lora_test_utils.py +++ b/tests/unittest/llmapi/lora_test_utils.py @@ -1,134 +1,86 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + import json import tarfile import tempfile +from contextlib import contextmanager from dataclasses import asdict, dataclass from pathlib import Path -from typing import List, Optional, OrderedDict, Tuple, Type, Union +from typing import Generator, List, Optional, Tuple, Union import pytest import torch +from safetensors.torch import save_file +from transformers import AutoConfig from utils.llm_data import llm_models_root -from utils.util import duplicate_list_to_length, flatten_list, similar -from tensorrt_llm import SamplingParams from tensorrt_llm._torch.peft.lora.cuda_graph_lora_params import \ CudaGraphLoraParams from tensorrt_llm._torch.peft.lora.layer import (GroupedGemmParamsInput, GroupedGemmParamsOutput, LoraLayer) -from tensorrt_llm.executor.request import LoRARequest -from tensorrt_llm.llmapi.llm import BaseLLM from tensorrt_llm.llmapi.llm_args import CudaGraphConfig from .test_utils import DelayedAssert - -def check_llama_7b_multi_unique_lora_adapters_from_request( - lora_adapter_count_per_call: List[int], repeat_calls: int, - repeats_per_call: int, llm_class: Type[BaseLLM], **llm_kwargs): - """Calls llm.generate s.t. for each C in lora_adapter_count_per_call, llm.generate is called with C requests - repeated 'repeats_per_call' times, where each request is configured with a unique LoRA adapter ID. - This entire process is done in a loop 'repeats_per_call' times with the same requests. - Asserts the output of each llm.generate call is similar to the expected. - """ # noqa: D205 - total_lora_adapters = sum(lora_adapter_count_per_call) - hf_model_dir = f"{llm_models_root()}/llama-models/llama-7b-hf" - hf_lora_dirs = [ - f"{llm_models_root()}/llama-models/luotuo-lora-7b-0.1", - f"{llm_models_root()}/llama-models/Japanese-Alpaca-LoRA-7b-v0" - ] - # Each prompt should have a reference for every LoRA adapter dir (in the same order as in hf_lora_dirs) - prompt_to_references = OrderedDict({ - "美国的首都在哪里? \n答案:": [ - "美国的首都是华盛顿。\n\n美国的", - "纽约\n\n### カンファレンスの", - ], - "アメリカ合衆国の首都はどこですか? \n答え:": [ - "华盛顿。\n\n英国の首都是什", - "ワシントン\nQ1. アメリカ合衆国", - ], - }) - - prompts_to_generate = duplicate_list_to_length( - flatten_list([[prompt] * len(hf_lora_dirs) - for prompt in prompt_to_references.keys()]), - total_lora_adapters) - references = duplicate_list_to_length( - flatten_list(list(prompt_to_references.values())), total_lora_adapters) - lora_requests = [ - LoRARequest(str(i), i, hf_lora_dirs[i % len(hf_lora_dirs)]) - for i in range(total_lora_adapters) - ] - llm = llm_class(hf_model_dir, **llm_kwargs) - - # Perform repeats of the same requests to test reuse and reload of adapters previously unloaded from cache - try: - for _ in range(repeat_calls): - last_idx = 0 - for adapter_count in lora_adapter_count_per_call: - sampling_params = SamplingParams(max_tokens=20) - outputs = llm.generate( - prompts_to_generate[last_idx:last_idx + adapter_count] * - repeats_per_call, - sampling_params, - lora_request=lora_requests[last_idx:last_idx + - adapter_count] * - repeats_per_call) - for output, ref in zip( - outputs, references[last_idx:last_idx + adapter_count] * - repeats_per_call): - assert similar(output.outputs[0].text, ref) - last_idx += adapter_count - finally: - llm.shutdown() - - -def check_llama_7b_multi_lora_from_request_test_harness( - llm_class: Type[BaseLLM], **llm_kwargs) -> None: - hf_model_dir = f"{llm_models_root()}/llama-models/llama-7b-hf" - hf_lora_dir1 = f"{llm_models_root()}/llama-models/luotuo-lora-7b-0.1" - hf_lora_dir2 = f"{llm_models_root()}/llama-models/Japanese-Alpaca-LoRA-7b-v0" - prompts = [ - "美国的首都在哪里? \n答案:", - "美国的首都在哪里? \n答案:", - "美国的首都在哪里? \n答案:", - "アメリカ合衆国の首都はどこですか? \n答え:", - "アメリカ合衆国の首都はどこですか? \n答え:", - "アメリカ合衆国の首都はどこですか? \n答え:", - ] - references = [ - "沃尔玛\n\n## 新闻\n\n* ", - "美国的首都是华盛顿。\n\n美国的", - "纽约\n\n### カンファレンスの", - "Washington, D.C.\nWashington, D.C. is the capital of the United", - "华盛顿。\n\n英国の首都是什", - "ワシントン\nQ1. アメリカ合衆国", - ] - key_words = [ - "沃尔玛", - "华盛顿", - "纽约", - "Washington", - "华盛顿", - "ワシントン", - ] - lora_req1 = LoRARequest("luotuo", 1, hf_lora_dir1) - lora_req2 = LoRARequest("Japanese", 2, hf_lora_dir2) - sampling_params = SamplingParams(max_tokens=20) - - llm = llm_class(hf_model_dir, **llm_kwargs) - try: - outputs = llm.generate(prompts, - sampling_params, - lora_request=[ - None, lora_req1, lora_req2, None, lora_req1, - lora_req2 - ]) - finally: - llm.shutdown() - for output, ref, key_word in zip(outputs, references, key_words): - assert similar(output.outputs[0].text, - ref) or key_word in output.outputs[0].text +QWEN3_MODEL_DIR = Path(llm_models_root()) / "Qwen3" / "Qwen3-0.6B" + + +def create_qwen3_lora_adapter(adapter_dir: Path, rank: int = 8) -> Path: + """Create a deterministic Qwen3 attention LoRA adapter for server tests.""" + config = AutoConfig.from_pretrained(QWEN3_MODEL_DIR) + head_dim = getattr(config, "head_dim", + config.hidden_size // config.num_attention_heads) + projection_output_sizes = { + "q_proj": config.num_attention_heads * head_dim, + "k_proj": config.num_key_value_heads * head_dim, + "v_proj": config.num_key_value_heads * head_dim, + } + generator = torch.Generator().manual_seed(42) + weights = {} + for module_name, output_size in projection_output_sizes.items(): + prefix = f"base_model.model.model.layers.0.self_attn.{module_name}" + weights[f"{prefix}.lora_A.weight"] = torch.randn( + rank, config.hidden_size, generator=generator) * 0.1 + weights[f"{prefix}.lora_B.weight"] = torch.randn( + output_size, rank, generator=generator) * 0.1 + + adapter_dir.mkdir(parents=True, exist_ok=True) + save_file(weights, adapter_dir / "adapter_model.safetensors") + adapter_config = { + "base_model_name_or_path": str(QWEN3_MODEL_DIR), + "bias": "none", + "inference_mode": True, + "lora_alpha": rank, + "lora_dropout": 0.0, + "peft_type": "LORA", + "r": rank, + "target_modules": list(projection_output_sizes), + "task_type": "CAUSAL_LM", + } + with open(adapter_dir / "adapter_config.json", "w", + encoding="utf-8") as config_file: + json.dump(adapter_config, config_file) + return adapter_dir + + +@contextmanager +def qwen3_lora_adapter() -> Generator[Path, None, None]: + with tempfile.TemporaryDirectory() as temp_dir: + yield create_qwen3_lora_adapter(Path(temp_dir) / "qwen3-lora") def create_mock_nemo_lora_checkpoint( diff --git a/tests/unittest/llmapi/test_llm.py b/tests/unittest/llmapi/test_llm.py index 70c438f0ffb3..5b62972ec8c2 100644 --- a/tests/unittest/llmapi/test_llm.py +++ b/tests/unittest/llmapi/test_llm.py @@ -1,3 +1,18 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + import asyncio import contextlib import datetime @@ -39,8 +54,6 @@ # isort: on # The unittests are based on the tiny-llama, which is fast to build and run. -# There are other tests based on llama-7B model, such as the end-to-end tests in test_e2e.py, and parallel tests in -# test_llm_multi_gpu.py. pytestmark = pytest.mark.threadleak(enabled=False) diff --git a/tests/unittest/llmapi/test_llm_multi_gpu_pytorch.py b/tests/unittest/llmapi/test_llm_multi_gpu_pytorch.py index c3db8d23573c..01b953fc00db 100644 --- a/tests/unittest/llmapi/test_llm_multi_gpu_pytorch.py +++ b/tests/unittest/llmapi/test_llm_multi_gpu_pytorch.py @@ -1,21 +1,19 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + import pytest from utils.util import skip_ray from tensorrt_llm import LLM -from tensorrt_llm._torch.peft.lora.config import LoraConfig from tensorrt_llm.executor.rpc_proxy import GenerationExecutorRpcProxy from tensorrt_llm.llmapi import KvCacheConfig from tensorrt_llm.sampling_params import SamplingParams -from .lora_test_utils import ( - check_llama_7b_multi_lora_from_request_test_harness, - test_lora_with_and_without_cuda_graph) from .test_llm import (_test_llm_capture_request_error, llama_model_path, llm_get_stats_async_test_harness, llm_get_stats_test_harness, llm_return_logprobs_test_harness, tinyllama_logits_processor_test_harness) -from .test_llm_pytorch import llama_7b_lora_from_dir_test_harness global_kv_cache_config = KvCacheConfig(free_gpu_memory_fraction=0.4) @@ -41,31 +39,6 @@ def test_tinyllama_logits_processor_2gpu(tp_size: int, pp_size: int): pipeline_parallel_size=pp_size) -@pytest.mark.gpu2 -def test_llama_7b_lora_tp2(): - llama_7b_lora_from_dir_test_harness(tensor_parallel_size=2, - kv_cache_config=global_kv_cache_config) - - -@pytest.mark.gpu4 -@skip_ray # https://nvbugs/5682551 -@test_lora_with_and_without_cuda_graph -def test_llama_7b_multi_lora_tp4(cuda_graph_config): - # For LoRA checkpoints without finetuned embedding and lm_head, we can either: - # (1) specify lora_target_modules, or - # (2) provide a lora_dir to infer the lora_target_modules. - lora_config = LoraConfig(lora_target_modules=['attn_q', 'attn_k', 'attn_v'], - max_lora_rank=8, - max_loras=1, - max_cpu_loras=8) - check_llama_7b_multi_lora_from_request_test_harness( - LLM, - lora_config=lora_config, - tensor_parallel_size=4, - kv_cache_config=global_kv_cache_config, - cuda_graph_config=cuda_graph_config) - - @skip_ray @pytest.mark.gpu2 def test_llm_rpc_tp2(): diff --git a/tests/unittest/llmapi/test_llm_pytorch.py b/tests/unittest/llmapi/test_llm_pytorch.py index 5dbe7e52351f..aa4ddd1b1c96 100644 --- a/tests/unittest/llmapi/test_llm_pytorch.py +++ b/tests/unittest/llmapi/test_llm_pytorch.py @@ -16,17 +16,16 @@ from tensorrt_llm.executor import GenerationExecutorWorker, RequestError from tensorrt_llm.executor.rpc_proxy import GenerationExecutorRpcProxy from tensorrt_llm.llmapi import CacheTransceiverConfig, KvCacheConfig -from tensorrt_llm.llmapi.llm_args import (NGramDecodingConfig, PeftCacheConfig, - SchedulerConfig, WaitingQueuePolicy) +from tensorrt_llm.llmapi.llm_args import (NGramDecodingConfig, SchedulerConfig, + WaitingQueuePolicy) from tensorrt_llm.metrics import MetricNames from tensorrt_llm.sampling_params import SamplingParams # isort: off -from .lora_test_utils import ( - check_llama_7b_multi_lora_from_request_test_harness, - check_llama_7b_multi_unique_lora_adapters_from_request, - create_mock_nemo_lora_checkpoint, compare_cuda_graph_lora_params_filler, - CUDAGraphLoRATestParams, test_lora_with_and_without_cuda_graph) +from .lora_test_utils import (create_mock_nemo_lora_checkpoint, + compare_cuda_graph_lora_params_filler, + CUDAGraphLoRATestParams, + test_lora_with_and_without_cuda_graph) from .test_llm import (_test_llm_capture_request_error, get_model_path, global_kvcache_config, global_kvcache_config_no_reuse, llama_model_path, llm_get_stats_async_test_harness, @@ -36,10 +35,9 @@ sampling_params_for_aborting_request, run_llm_with_postprocess_parallel_and_result_handler, tinyllama_logits_processor_test_harness) -from utils.util import (force_ampere, similar, similarity_score, - skip_fp8_pre_ada, skip_gpu_memory_less_than_40gb, - skip_gpu_memory_less_than_80gb, - skip_gpu_memory_less_than_138gb, skip_ray) +from utils.util import (force_ampere, similar, skip_fp8_pre_ada, + skip_gpu_memory_less_than_40gb, + skip_gpu_memory_less_than_80gb, skip_ray) from utils.llm_data import llm_models_root from tensorrt_llm._torch.peft.lora.config import LoraConfig from tensorrt_llm.executor.request import LoRARequest @@ -380,272 +378,6 @@ def test_lora_cuda_graph_params_filling_kernel_special_cases(): compare_cuda_graph_lora_params_filler(test_params6) -def llama_7b_lora_from_dir_test_harness(**llm_kwargs) -> None: - lora_config = LoraConfig( - lora_dir=[f"{llm_models_root()}/llama-models/luotuo-lora-7b-0.1"], - max_lora_rank=8, - max_loras=2, - max_cpu_loras=2) - llm = LLM(model=f"{llm_models_root()}/llama-models/llama-7b-hf", - lora_config=lora_config, - **llm_kwargs) - try: - prompts = [ - "美国的首都在哪里? \n答案:", - ] - references = [ - "美国的首都是华盛顿。\n\n美国的", - ] - sampling_params = SamplingParams(max_tokens=20) - lora_req = LoRARequest( - "task-0", 0, f"{llm_models_root()}/llama-models/luotuo-lora-7b-0.1") - lora_request = [lora_req] - - outputs = llm.generate(prompts, - sampling_params, - lora_request=lora_request) - assert similar(outputs[0].outputs[0].text, references[0]) - finally: - llm.shutdown() - - -@skip_gpu_memory_less_than_40gb -@pytest.mark.part0 -@test_lora_with_and_without_cuda_graph -@pytest.mark.parametrize("use_speculative", [True, False]) -def test_llama_7b_lora(cuda_graph_config, use_speculative): - llm_kwargs = { - "cuda_graph_config": - cuda_graph_config, - "speculative_config": - NGramDecodingConfig(max_draft_len=5) if use_speculative else None - } - llama_7b_lora_from_dir_test_harness(**llm_kwargs) - - -@skip_gpu_memory_less_than_40gb -@test_lora_with_and_without_cuda_graph -@pytest.mark.parametrize("use_speculative", [True, False]) -def test_llama_7b_lora_default_modules(cuda_graph_config, - use_speculative) -> None: - lora_config = LoraConfig(max_lora_rank=64, max_loras=2, max_cpu_loras=2) - - hf_model_dir = f"{llm_models_root()}/llama-models/llama-7b-hf" - - llm = LLM(model=hf_model_dir, - lora_config=lora_config, - speculative_config=NGramDecodingConfig( - max_draft_len=5) if use_speculative else None, - cuda_graph_config=cuda_graph_config) - - hf_lora_dir = f"{llm_models_root()}/llama-models/luotuo-lora-7b-0.1" - try: - prompts = [ - "美国的首都在哪里? \n答案:", - ] - references = [ - "美国的首都是华盛顿。\n\n美国的", - ] - sampling_params = SamplingParams(max_tokens=20, - add_special_tokens=False) - lora_req = LoRARequest("luotuo", 1, hf_lora_dir) - lora_request = [lora_req] - - outputs = llm.generate(prompts, - sampling_params, - lora_request=lora_request) - - assert similar(outputs[0].outputs[0].text, references[0]) - finally: - llm.shutdown() - - -def _check_llama_7b_multi_lora_evict_load_new_adapters( - lora_adapter_count_per_call: list[int], max_loras: int, - max_cpu_loras: int, repeat_calls: int, repeats_per_call: int, - **llm_kwargs): - # For LoRA checkpoints without finetuned embedding and lm_head, we can either: - # (1) specify lora_target_modules, or - # (2) provide a lora_dir to infer the lora_target_modules. - lora_config = LoraConfig(lora_target_modules=['attn_q', 'attn_k', 'attn_v'], - max_lora_rank=8, - max_loras=max_loras, - max_cpu_loras=max_cpu_loras) - check_llama_7b_multi_unique_lora_adapters_from_request( - lora_adapter_count_per_call, - repeat_calls, - repeats_per_call, - LLM, - lora_config=lora_config, - **llm_kwargs) - - -@skip_gpu_memory_less_than_40gb -@skip_ray # https://nvbugs/5682551 -@pytest.mark.part3 -@test_lora_with_and_without_cuda_graph -def test_llama_7b_multi_lora_evict_and_reload_lora_gpu_cache(cuda_graph_config): - """Test eviction and re-loading a previously evicted adapter from the LoRA GPU cache, within a single - llm.generate call, that's repeated twice. - """ # noqa: D205 - _check_llama_7b_multi_lora_evict_load_new_adapters( - lora_adapter_count_per_call=[2], - max_loras=1, - max_cpu_loras=2, - repeat_calls=2, - repeats_per_call=3, - cuda_graph_config=cuda_graph_config) - - -@skip_gpu_memory_less_than_40gb -@pytest.mark.part1 -@test_lora_with_and_without_cuda_graph -def test_llama_7b_multi_lora_evict_and_load_new_adapters_in_cpu_and_gpu_cache( - cuda_graph_config): - """Test eviction and loading of new adapters in the evicted space, over several llm.generate calls, with LoRA GPU - cache size < LoRA CPU cache size. - """ # noqa: D205 - _check_llama_7b_multi_lora_evict_load_new_adapters( - lora_adapter_count_per_call=[2, 2, 2], - max_loras=1, - max_cpu_loras=3, - repeat_calls=1, - repeats_per_call=1, - cuda_graph_config=cuda_graph_config) - - -@skip_gpu_memory_less_than_40gb -@pytest.mark.part0 -@test_lora_with_and_without_cuda_graph -def test_llama_7b_multi_lora_read_from_cache_after_insert(cuda_graph_config): - """Test that loading and then using the same adapters loaded in cache works.""" - _check_llama_7b_multi_lora_evict_load_new_adapters( - lora_adapter_count_per_call=[3], - max_loras=3, - max_cpu_loras=3, - repeat_calls=2, - repeats_per_call=1, - cuda_graph_config=cuda_graph_config) - - -@skip_gpu_memory_less_than_40gb -@pytest.mark.part3 -@test_lora_with_and_without_cuda_graph -def test_llama_7b_multi_lora_evict_and_reload_evicted_adapters_in_cpu_and_gpu_cache( - cuda_graph_config): - """Test eviction, reloading new adapters and reloading previously evicted adapters from the LoRA CPU cache & GPU - cache over multiple llm.generate call repeated twice (two calls with the same requests): - At the end of the 1st llm.generate call: - The LoRA caches should contain adapters 1, 2 and shouldn't contain adapter 0 (it should have been evicted). - So in the 2nd call, the worker should: - - Send req0 with adapter 0 weights (because it was previously evicted) - - Send the other two requests without their adapter weights as they're already in LoRA CPU cache - Then, handling of req0 that has weights but not in the cache should evict one of the other two adapters from - the cache, causing that evicted adapter's request to again load its weights from the file system, as they - aren't with the request and aren't in LoRA cache. - """ # noqa: D205 - _check_llama_7b_multi_lora_evict_load_new_adapters( - lora_adapter_count_per_call=[3], - max_loras=2, - max_cpu_loras=2, - repeat_calls=2, - repeats_per_call=1, - cuda_graph_config=cuda_graph_config) - - -@skip_gpu_memory_less_than_40gb -@pytest.mark.part2 -@test_lora_with_and_without_cuda_graph -def test_llama_7b_peft_cache_config_affects_peft_cache_size(cuda_graph_config): - """Tests that LLM arg of peft_cache_config affects the peft cache sizes. - - NOTE: The caller can't get the actual LoRA cache sizes, so we instead we - test that it fails when configured with a value too small to contain a - single adapter. - """ - # For LoRA checkpoints without finetuned embedding and lm_head, we can either: - # (1) specify lora_target_modules, or - # (2) provide a lora_dir to infer the lora_target_modules. - lora_config_no_cache_size_values = LoraConfig( - lora_target_modules=['attn_q', 'attn_k', 'attn_v'], max_lora_rank=8) - - # Test that too small PeftCacheConfig.host_cache_size causes failure - with pytest.raises(RuntimeError): - check_llama_7b_multi_lora_from_request_test_harness( - LLM, - lora_config=lora_config_no_cache_size_values, - peft_cache_config=PeftCacheConfig( - host_cache_size=1), # size in bytes - cuda_graph_config=cuda_graph_config) - - # Test that too small PeftCacheConfig.device_cache_percent causes failure - with pytest.raises(RuntimeError): - check_llama_7b_multi_lora_from_request_test_harness( - LLM, - lora_config=lora_config_no_cache_size_values, - peft_cache_config=PeftCacheConfig(device_cache_percent=0.0000001), - cuda_graph_config=cuda_graph_config) - - -@skip_ray # https://nvbugs/5682551 -@skip_gpu_memory_less_than_40gb -# https://nvbugs/6566707: hung for 2400s in late executor-init/first-generate -# on a many-times-reused MPI pool; isolate on a private pool until root-caused. -@pytest.mark.private_mpi_session -@pytest.mark.part1 -@test_lora_with_and_without_cuda_graph -def test_llama_7b_lora_config_overrides_peft_cache_config(cuda_graph_config): - """Tests that cache size args in lora_config LLM arg override the cache size - parameters in peft_cache_config LLM arg. - """ # noqa: D205 - check_llama_7b_multi_lora_from_request_test_harness( - LLM, - lora_config=LoraConfig( - lora_target_modules=['attn_q', 'attn_k', 'attn_v'], - max_lora_rank=8, - max_loras=2, - max_cpu_loras=2), - peft_cache_config=PeftCacheConfig( - host_cache_size=1, # size in bytes - device_cache_percent=0.0000001), - cuda_graph_config=cuda_graph_config) - - -@skip_gpu_memory_less_than_138gb -@pytest.mark.part1 -@test_lora_with_and_without_cuda_graph -def test_nemotron_nas_lora(cuda_graph_config) -> None: - lora_config = LoraConfig(lora_dir=[ - f"{llm_models_root()}/nemotron-nas/Llama-3_3-Nemotron-Super-49B-v1-lora-adapter_r64" - ], - max_lora_rank=64, - max_loras=1, - max_cpu_loras=1) - - llm = LLM( - model= - f"{llm_models_root()}/nemotron-nas/Llama-3_3-Nemotron-Super-49B-v1", - lora_config=lora_config, - cuda_graph_config=cuda_graph_config, - trust_remote_code=True) - - prompts = [ - "Hello, how are you?", - "Hello, how are you?", - ] - - sampling_params = SamplingParams(max_tokens=3, add_special_tokens=False) - lora_req = LoRARequest( - "task-0", 0, - f"{llm_models_root()}/nemotron-nas/Llama-3_3-Nemotron-Super-49B-v1-lora-adapter_r64" - ) - lora_request = [lora_req, None] - - outputs = llm.generate(prompts, sampling_params, lora_request=lora_request) - - assert similar(outputs[0].outputs[0].text, outputs[1].outputs[0].text) - - @skip_gpu_memory_less_than_80gb @pytest.mark.part0 @test_lora_with_and_without_cuda_graph @@ -675,42 +407,6 @@ def test_llama_3_1_8b_fp8_with_bf16_lora(cuda_graph_config) -> None: assert similar(output.outputs[0].text, reference) -@skip_ray # https://nvbugs/5682551 -@skip_gpu_memory_less_than_80gb -def test_llama_3_3_70b_fp8_with_squad_lora_tp2() -> None: - skip_fp8_pre_ada(use_fp8=True) - - model_dir = f"{llm_models_root()}/llama-3.3-models/Llama-3.3-70B-Instruct-FP8" - lora_dir = f"{llm_models_root()}/llama-3.3-models/Llama-3.3-70B-Instruct-FP8-lora-adapter_NIM_r8" - - prompt = "What is the capital of the United States?" - expected_output = " Washington, D.C.\nWhat is the capital of the United States? Washington, D.C." - - lora_config = LoraConfig(lora_dir=[lora_dir], - max_lora_rank=8, - max_loras=2, - max_cpu_loras=2) - lora_req = LoRARequest("squad-lora", 0, lora_dir) - - llm = LLM(model_dir, - tensor_parallel_size=2, - lora_config=lora_config, - cuda_graph_config=None) - - try: - output = llm.generate(prompt, - SamplingParams(max_tokens=50, temperature=0.0), - lora_request=[lora_req]) - generated_text = output.outputs[0].text - print(f"Generated output: {repr(generated_text)}") - - similarity = similarity_score(generated_text, expected_output) - assert similar(generated_text, expected_output, threshold=0.8), \ - f"Output similarity too low (similarity={similarity:.2%})!\nExpected: {repr(expected_output)}\nGot: {repr(generated_text)}" - finally: - llm.shutdown() - - @pytest.mark.part2 @test_lora_with_and_without_cuda_graph def test_gemma3_1b_instruct_multi_lora(cuda_graph_config) -> None: diff --git a/tests/unittest/llmapi/test_memory_profiling.py b/tests/unittest/llmapi/test_memory_profiling.py index 264c5b29c900..c590378cc52f 100644 --- a/tests/unittest/llmapi/test_memory_profiling.py +++ b/tests/unittest/llmapi/test_memory_profiling.py @@ -1,3 +1,18 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + import pytest import torch @@ -82,8 +97,7 @@ def test_pyexecutor_and_kvcache_share_execution_stream(): Both components must use the same stream for proper synchronization. """ - # Use a simple model for testing - MODEL = "llama-3.2-models/Llama-3.2-1B-Instruct" + MODEL = "Qwen3/Qwen3-0.6B" MODEL_PATH = get_model_path(MODEL) kv_cache_config = KvCacheConfig(enable_block_reuse=False, diff --git a/tests/unittest/metrics/test_collector.py b/tests/unittest/metrics/test_collector.py index c42f5eab10b1..7a9558ba3864 100644 --- a/tests/unittest/metrics/test_collector.py +++ b/tests/unittest/metrics/test_collector.py @@ -217,8 +217,8 @@ class TestConfigInfoMetrics: def test_model_config_info(self, collector): model_config = { - "model": "meta-llama/Llama-3-8B", - "served_model_name": "Llama-3-8B", + "model": "meta-llama/Llama-3.1-8B-Instruct", + "served_model_name": "Llama-3.1-8B-Instruct", "dtype": "float16", "quantization": "none", "max_model_len": "4096", diff --git a/tests/unittest/tools/test_config_selector.js b/tests/unittest/tools/test_config_selector.js index ce489982cc69..95240487bdf1 100644 --- a/tests/unittest/tools/test_config_selector.js +++ b/tests/unittest/tools/test_config_selector.js @@ -12,7 +12,6 @@ const SELECTOR_JS = path.join(REPO_ROOT, "docs/source/_static/config_selector.js const CONFIG_DB_JSON = path.join(REPO_ROOT, "docs/source/_static/config_db.json"); const DEEPSEEK_MODEL = "deepseek-ai/DeepSeek-R1-0528"; const DEEPSEEK_NVFP4_MODEL = "nvidia/DeepSeek-R1-0528-FP4-v2"; -const LLAMA_FP8_MODEL = "nvidia/Llama-3.3-70B-Instruct-FP8"; function loadSelectorExports() { const source = fs.readFileSync(SELECTOR_JS, "utf8"); @@ -374,12 +373,3 @@ test("data-models filter applies to curated entries too", () => { "gpt-oss should be filtered out" ); }); - -test("Llama-3.3-70B has exactly one curated entry", () => { - const selector = loadSelectorExports(); - const curated = loadCuratedEntries(); - - const llamaCurated = selector.curatedEntriesForModel(curated, LLAMA_FP8_MODEL); - assert.equal(llamaCurated.length, 1, "Llama-3.3-70B should have exactly 1 curated entry"); - assert.equal(llamaCurated[0].scenario, "Max Throughput"); -});