Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 0 additions & 2 deletions tests/integration/defs/accuracy/references/mmlu.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -202,8 +202,6 @@ deepseek-ai/DeepSeek-R1-0528:
- quant_algo: FP8_BLOCK_SCALES
kv_cache_quant_algo: FP8
accuracy: 84.722
deepseek-ai/DeepSeek-R1-Distill-Llama-70B:
- accuracy: 78.19
deepseek-ai/DeepSeek-V3.2-Exp:
- quant_algo: FP8_BLOCK_SCALES
accuracy: 88.2
Expand Down
14 changes: 0 additions & 14 deletions tests/integration/defs/accuracy/test_llm_api_pytorch.py
Original file line number Diff line number Diff line change
Expand Up @@ -3306,20 +3306,6 @@ def test_fp8_blockscale_chunked_prefill(self, tp_size, pp_size, ep_size,
task.evaluate(llm)


class TestDeepSeekR1DistillLlama70B(LlmapiAccuracyTestHarness):
MODEL_NAME = "deepseek-ai/DeepSeek-R1-Distill-Llama-70B"
MODEL_PATH = f"{llm_models_root()}/DeepSeek-R1/DeepSeek-R1-Distill-Llama-70B"

@skip_pre_hopper
@pytest.mark.skip_less_mpi_world_size(2)
def test_auto_dtype_tp2(self):
kv_cache_config = KvCacheConfig(free_gpu_memory_fraction=0.4)
_run_multinode_accuracy(self.MODEL_PATH,
self.MODEL_NAME,
benchmarks=["mmlu"],
kv_cache_config=kv_cache_config)


@pytest.mark.timeout(14400)
@pytest.mark.skip_less_mpi_world_size(8)
class TestDeepSeekV3(LlmapiAccuracyTestHarness):
Expand Down
2 changes: 0 additions & 2 deletions tests/integration/defs/perf/_model_paths.py
Original file line number Diff line number Diff line change
Expand Up @@ -26,8 +26,6 @@
"llama_v4_scout_17b_16e_instruct": "llama4-models/Llama-4-Scout-17B-16E-Instruct",
"llama_v4_scout_17b_16e_instruct_fp8": "llama4-models/Llama-4-Scout-17B-16E-Instruct-FP8",
"llama_v4_scout_17b_16e_instruct_fp4": "llama4-models/Llama-4-Scout-17B-16E-Instruct-FP4",
"deepseek_r1_distill_qwen_32b": "DeepSeek-R1/DeepSeek-R1-Distill-Qwen-32B",
"deepseek_r1_distill_llama_70b": "DeepSeek-R1/DeepSeek-R1-Distill-Llama-70B/",
"gemma_3_27b_it": "gemma/gemma-3-27b-it",
"gemma_3_27b_it_fp8": "gemma/gemma-3-27b-it-fp8",
"gemma_3_27b_it_fp4": "gemma/gemma-3-27b-it-FP4",
Expand Down
4 changes: 0 additions & 4 deletions tests/integration/defs/perf/base_perf_pytorch.csv
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,3 @@ network_name,perf_case_name,test_name,threshold,absolute_threshold,metric_type,p
"llama_v3.1_8b_instruct-bench-pytorch-float16-maxbs:512-maxnt:2048-input_output_len:128,128-reqs:8192","H100_PCIe-PyTorch-Perf-1/perf/test_perf.py::test_perf_metric_seq_throughput[llama_v3.1_8b_instruct-bench-pytorch-float16-maxbs:512-maxnt:2048-input_output_len:128,128-reqs:8192]","test_perf_metric_seq_throughput[llama_v3.1_8b_instruct-bench-pytorch-float16-maxbs:512-maxnt:2048-input_output_len:128,128-reqs:8192]",-0.20,5,SEQ_THROUGHPUT,76.45,
"llama_v3.1_8b_instruct-bench-pytorch-float16-maxbs:512-maxnt:2048-input_output_len:128,128-reqs:8192","H100_PCIe-PyTorch-Perf-1/perf/test_perf.py::test_perf_metric_total_output_throughput[llama_v3.1_8b_instruct-bench-pytorch-float16-maxbs:512-maxnt:2048-input_output_len:128,128-reqs:8192]","test_perf_metric_total_output_throughput[llama_v3.1_8b_instruct-bench-pytorch-float16-maxbs:512-maxnt:2048-input_output_len:128,128-reqs:8192]",-0.20,500,TOTAL_OUTPUT_THROUGHPUT,9785.75,
"llama_v3.1_8b_instruct-bench-pytorch-float16-maxbs:512-maxnt:2048-input_output_len:128,128-reqs:8192","H100_PCIe-PyTorch-Perf-1/perf/test_perf.py::test_perf_metric_kv_cache_size[llama_v3.1_8b_instruct-bench-pytorch-float16-maxbs:512-maxnt:2048-input_output_len:128,128-reqs:8192]","test_perf_metric_kv_cache_size[llama_v3.1_8b_instruct-bench-pytorch-float16-maxbs:512-maxnt:2048-input_output_len:128,128-reqs:8192]",0.20,2,KV_CACHE_SIZE,55.64,
"deepseek_r1_distill_qwen_32b-bench-_autodeploy-float16-maxbs:512-maxnt:2048-kv_frac:0.8-input_output_len:1024,1024","H100_PCIe-PyTorch-Perf-1/perf/test_perf.py::test_perf_metric_inference_time[deepseek_r1_distill_qwen_32b-subtype:H100_PCIe-bench-_autodeploy-float16-maxbs:512-maxnt:2048-kv_frac:0.8-input_output_len:1024,1024]","test_perf_metric_inference_time[deepseek_r1_distill_qwen_32b-subtype:H100_PCIe-bench-_autodeploy-float16-maxbs:512-maxnt:2048-kv_frac:0.8-input_output_len:1024,1024]",0.1,50,INFERENCE_TIME,1359184.5059,H100_PCIe
"deepseek_r1_distill_qwen_32b-bench-_autodeploy-float16-maxbs:512-maxnt:2048-kv_frac:0.8-input_output_len:1024,1024","H100_PCIe-PyTorch-Perf-1/perf/test_perf.py::test_perf_metric_kv_cache_size[deepseek_r1_distill_qwen_32b-subtype:H100_PCIe-bench-_autodeploy-float16-maxbs:512-maxnt:2048-kv_frac:0.8-input_output_len:1024,1024]","test_perf_metric_kv_cache_size[deepseek_r1_distill_qwen_32b-subtype:H100_PCIe-bench-_autodeploy-float16-maxbs:512-maxnt:2048-kv_frac:0.8-input_output_len:1024,1024]",-0.1,50,KV_CACHE_SIZE,10.92,H100_PCIe
"deepseek_r1_distill_qwen_32b-bench-_autodeploy-float16-maxbs:512-maxnt:2048-kv_frac:0.8-input_output_len:1024,1024","H100_PCIe-PyTorch-Perf-1/perf/test_perf.py::test_perf_metric_seq_throughput[deepseek_r1_distill_qwen_32b-subtype:H100_PCIe-bench-_autodeploy-float16-maxbs:512-maxnt:2048-kv_frac:0.8-input_output_len:1024,1024]","test_perf_metric_seq_throughput[deepseek_r1_distill_qwen_32b-subtype:H100_PCIe-bench-_autodeploy-float16-maxbs:512-maxnt:2048-kv_frac:0.8-input_output_len:1024,1024]",-0.1,10,SEQ_THROUGHPUT,0.3767,H100_PCIe
"deepseek_r1_distill_qwen_32b-bench-_autodeploy-float16-maxbs:512-maxnt:2048-kv_frac:0.8-input_output_len:1024,1024","H100_PCIe-PyTorch-Perf-1/perf/test_perf.py::test_perf_metric_total_output_throughput[deepseek_r1_distill_qwen_32b-subtype:H100_PCIe-bench-_autodeploy-float16-maxbs:512-maxnt:2048-kv_frac:0.8-input_output_len:1024,1024]","test_perf_metric_total_output_throughput[deepseek_r1_distill_qwen_32b-subtype:H100_PCIe-bench-_autodeploy-float16-maxbs:512-maxnt:2048-kv_frac:0.8-input_output_len:1024,1024]",-0.1,10,TOTAL_OUTPUT_THROUGHPUT,385.7372,H100_PCIe
10 changes: 0 additions & 10 deletions tests/integration/defs/perf/pytorch_model_config.py
Original file line number Diff line number Diff line change
Expand Up @@ -291,16 +291,6 @@ def get_model_yaml_config(model_label: str,
},
}
},
# Model-specific cases with attention_dp disabled to prevent hangs
{
'patterns': [
'deepseek_r1_distill_llama_70b',
],
'config': {
# True causes hang, needs model-specific fix.
'enable_attention_dp': False,
}
},
# Qwen3 models with fp4 quantization on B200 and fp8 quantization on H200/H20
{
'patterns': [
Expand Down
34 changes: 10 additions & 24 deletions tests/integration/defs/test_e2e.py
Original file line number Diff line number Diff line change
Expand Up @@ -41,13 +41,13 @@


@pytest.mark.parametrize("model_name,model_path", [
Comment thread
xinhe-nv marked this conversation as resolved.
("DeepSeek-R1-Distill-Qwen-1.5B", "DeepSeek-R1-Distill-Qwen-1.5B"),
("Qwen3/Qwen3-0.6B", "Qwen3/Qwen3-0.6B"),
])
def test_qwen_e2e_cpprunner_large_new_tokens(model_name, model_path, llm_venv):
"""RCCA: https://nvbugs/5238105 - none of n>1 sequences may be empty."""
from tensorrt_llm import LLM, SamplingParams

Comment thread
xinhe-nv marked this conversation as resolved.
prompt = r"<|begin▁of▁sentence|><|User|>The operation $\otimes$ is defined for all nonzero numbers by $a \otimes b = \frac{a^{2}}{b}$. Determine $[(1 \otimes 2) \otimes 3] - [1 \otimes (2 \otimes 3)]$. Let's think step by step and output the final answer within \boxed{}.<|Assistant|>"
prompt = r"The operation $\otimes$ is defined for all nonzero numbers by $a \otimes b = \frac{a^{2}}{b}$. Determine $[(1 \otimes 2) \otimes 3] - [1 \otimes (2 \otimes 3)]$. Let's think step by step."

sampling_params = SamplingParams(
max_tokens=1024,
Expand All @@ -60,6 +60,14 @@ def test_qwen_e2e_cpprunner_large_new_tokens(model_name, model_path, llm_venv):
with LLM(model=f"{llm_models_root()}/{model_path}",
max_batch_size=8,
max_seq_len=4224) as llm:
prompt = llm.tokenizer.apply_chat_template(
[{
"role": "user",
"content": prompt
}],
tokenize=False,
add_generation_prompt=True,
)
outputs = llm.generate([prompt], sampling_params=sampling_params)

completions = outputs[0].outputs
Expand Down Expand Up @@ -918,9 +926,6 @@ def test_ptp_quickstart(llm_root, llm_venv):
pytest.param('Mistral-Nemo-12b-Base',
'Mistral-Nemo-Base-2407',
marks=skip_pre_blackwell),
pytest.param('DeepSeek-R1-Distill-Qwen-32B',
'DeepSeek-R1/DeepSeek-R1-Distill-Qwen-32B',
marks=skip_pre_blackwell),
pytest.param('GPT-OSS-20B', 'gpt_oss/gpt-oss-20b',
marks=skip_pre_blackwell),
pytest.param(
Expand Down Expand Up @@ -1854,23 +1859,6 @@ def test_ptp_quickstart_bert(llm_root, llm_venv, model_name, model_path,
print("Success: HF model logits match TRTLLM logits!")


@pytest.mark.skip_less_device_memory(80000)
@pytest.mark.parametrize("model_name,model_path", [
("DeepSeek-R1-Distill-Qwen-7B", "DeepSeek-R1/DeepSeek-R1-Distill-Qwen-7B"),
])
def test_ptp_scaffolding(llm_root, llm_venv, model_name, model_path):
print(f"Testing scaffolding {model_name}.")
example_root = Path(os.path.join(llm_root, "examples", "scaffolding"))
input_file = Path(os.path.join(example_root, "test.jsonl"))
llm_venv.run_cmd([
str(example_root / "run_majority_vote_aime24.py"),
"--model_dir",
f"{llm_models_root()}/{model_path}",
f"--jsonl_file={input_file}",
"--threshold=0.5",
])


@pytest.mark.timeout(5400)
@pytest.mark.skip_less_device_memory(80000)
@pytest.mark.skip_less_device(4)
Expand Down Expand Up @@ -1933,8 +1921,6 @@ def test_multi_nodes_eval(model_path, tp_size, pp_size, ep_size, eval_task,
marks=skip_pre_hopper),
pytest.param('Qwen3/saved_models_Qwen3-235B-A22B_nvfp4_hf',
marks=skip_pre_blackwell),
pytest.param('DeepSeek-R1/DeepSeek-R1-Distill-Llama-70B',
marks=skip_pre_hopper),
pytest.param('llama4-models/Llama-4-Scout-17B-16E-Instruct-FP8',
marks=skip_pre_hopper),
pytest.param('llama4-models/Llama-4-Scout-17B-16E-Instruct',
Expand Down
Loading
Loading