diff --git a/tensorrt_llm/_torch/pyexecutor/sampler/sampler_strategy.py b/tensorrt_llm/_torch/pyexecutor/sampler/sampler_strategy.py index 51b69c1d4c24..475e49acd952 100644 --- a/tensorrt_llm/_torch/pyexecutor/sampler/sampler_strategy.py +++ b/tensorrt_llm/_torch/pyexecutor/sampler/sampler_strategy.py @@ -328,7 +328,7 @@ def sample( ) case ("greedy", None): tokens, softmax = greedy_search_sampling_batch(logits, return_probs=return_probs) - temperature = None + temperature = None # type: ignore[assignment] case ( "beam_search", beam_width_in, diff --git a/tests/integration/defs/accuracy/references/SlimPajama-6B.yaml b/tests/integration/defs/accuracy/references/SlimPajama-6B.yaml index 421d0f38de81..0967ef424bce 100644 --- a/tests/integration/defs/accuracy/references/SlimPajama-6B.yaml +++ b/tests/integration/defs/accuracy/references/SlimPajama-6B.yaml @@ -1,2 +1 @@ -gradientai/Llama-3-8B-Instruct-Gradient-1048k: - - accuracy: 7.663 +{} diff --git a/tests/integration/defs/accuracy/references/cnn_dailymail.yaml b/tests/integration/defs/accuracy/references/cnn_dailymail.yaml index b9f82bfe735b..6ae2f52519b8 100644 --- a/tests/integration/defs/accuracy/references/cnn_dailymail.yaml +++ b/tests/integration/defs/accuracy/references/cnn_dailymail.yaml @@ -70,32 +70,6 @@ lmsys/vicuna-7b-v1.3: accuracy: 27.832 - spec_dec_algo: Eagle3 accuracy: 27.832 -TinyLlama/TinyLlama-1.1B-Chat-v1.0: - - accuracy: 28.328 - - dtype: float32 - accuracy: 28.082 - - quant_algo: W8A16 - accuracy: 28.003 - - quant_algo: W8A16 - kv_cache_quant_algo: INT8 - accuracy: 27.089 - - quant_algo: W4A16 - accuracy: 25.194 - - quant_algo: W4A16 - kv_cache_quant_algo: INT8 - accuracy: 23.987 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 27.882 - - extra_acc_spec: pp_size=4 - accuracy: 15.123 -meta-llama/Meta-Llama-3-8B-Instruct: - - accuracy: 34.957 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 34.737 - - quant_algo: W8A16_GPTQ - accuracy: 34.858 meta-llama/Llama-3.1-8B: - accuracy: 24.360 - quant_algo: W8A8_SQ_PER_CHANNEL_PER_TOKEN_PLUGIN @@ -140,48 +114,6 @@ meta-llama/Llama-3.1-8B-Instruct: - quant_algo: FP8 extra_acc_spec: beam_width=2 accuracy: 31.201 -meta-llama/Llama-3.2-1B: - - accuracy: 27.427 - - quant_algo: W8A8_SQ_PER_CHANNEL_PER_TOKEN_PLUGIN - accuracy: 27.931 - - quant_algo: W8A8_SQ_PER_CHANNEL - accuracy: 25.631 - - quant_algo: W4A16_AWQ - accuracy: 25.028 - - quant_algo: W4A16_AWQ - kv_cache_quant_algo: INT8 - accuracy: 24.354 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 27.029 - - quant_algo: FP8 - accuracy: 27.029 - - quant_algo: FP8_PER_CHANNEL_PER_TOKEN - accuracy: 27.257 - - quant_algo: FP8_PER_CHANNEL_PER_TOKEN - extra_acc_spec: meta_recipe - accuracy: 27.614 - - extra_acc_spec: max_attention_window_size=960 - accuracy: 27.259 - - extra_acc_spec: max_attention_window_size=960;beam_width=4 - accuracy: 0 -meta-llama/Llama-3.2-3B: - - accuracy: 25.495 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 33.629 -meta-llama/Llama-3.3-70B-Instruct: - - quant_algo: FP8 - spec_dec_algo: Eagle - accuracy: 33.244 - - quant_algo: FP8 - spec_dec_algo: Eagle3 - accuracy: 33.244 - - quant_algo: NVFP4 - kv_cache_quant_algo: FP8 - accuracy: 34.383 - - quant_algo: FP8 - accuracy: 34.927 mistralai/Mistral-7B-v0.1: - accuracy: 25.741 - extra_acc_spec: beam_width=4 @@ -352,5 +284,3 @@ Qwen3/Qwen3-8B: - quant_algo: FP8_BLOCK_SCALES accuracy: 30 - accuracy: 30 -nvidia/Llama-3_3-Nemotron-Super-49B-v1: - - accuracy: 34.003 diff --git a/tests/integration/defs/accuracy/references/gpqa_diamond.yaml b/tests/integration/defs/accuracy/references/gpqa_diamond.yaml index 9483e8a43a40..d8adf5cd1c85 100644 --- a/tests/integration/defs/accuracy/references/gpqa_diamond.yaml +++ b/tests/integration/defs/accuracy/references/gpqa_diamond.yaml @@ -1,16 +1,3 @@ -meta-llama/Llama-3.3-70B-Instruct: - - accuracy: 45.96 - - quant_algo: NVFP4 - kv_cache_quant_algo: FP8 - accuracy: 45.55 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 48.03 - - quant_algo: FP8 - accuracy: 48.03 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 48.03 deepseek-ai/DeepSeek-R1: - quant_algo: NVFP4 accuracy: 70.45 @@ -35,30 +22,6 @@ deepseek-ai/DeepSeek-V3.2-Exp: - quant_algo: NVFP4 spec_dec_algo: MTP accuracy: 80.0 -nvidia/Llama-3_3-Nemotron-Super-49B-v1: - - accuracy: 44.95 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 42.42 - # GPQA diamond only contains 198 samples, so the score tends to have large variance. - # We repeated evaluation 7 times to choose a lower bound score for FP8, 42.42. - # random_seed=0: 47.98 - # random_seed=1: 42.42 - # random_seed=2: 52.02 - # random_seed=3: 51.52 - # random_seed=4: 48.48 - # random_seed=5: 47.47 - # random_seed=6: 45.96 -nvidia/Llama-3.1-Nemotron-Nano-8B-v1: - - accuracy: 40.40 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 39.39 -nvidia/Llama-3_1-Nemotron-Ultra-253B-v1: - - accuracy: 58.08 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 57.07 GPT-OSS/120B-MXFP4: - accuracy: 65.0 - spec_dec_algo: Eagle diff --git a/tests/integration/defs/accuracy/references/gsm8k.yaml b/tests/integration/defs/accuracy/references/gsm8k.yaml index 9f62b2bc501e..60f87453839a 100644 --- a/tests/integration/defs/accuracy/references/gsm8k.yaml +++ b/tests/integration/defs/accuracy/references/gsm8k.yaml @@ -31,24 +31,6 @@ meta-llama/Llama-3.1-8B-Instruct: - quant_algo: NVFP4 kv_cache_quant_algo: FP8 accuracy: 66.03 -meta-llama/Llama-3.3-70B-Instruct: - - accuracy: 83.78 - - quant_algo: NVFP4 - kv_cache_quant_algo: FP8 - accuracy: 87.33 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 90.30 - - quant_algo: FP8 - accuracy: 90.30 -meta-llama/Llama-4-Scout-17B-16E-Instruct: - - accuracy: 89.70 - - quant_algo: NVFP4 - kv_cache_quant_algo: FP8 - accuracy: 88.61 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 89.45 deepseek-ai/DeepSeek-V3-Lite: - accuracy: 64.74 - quant_algo: NVFP4 @@ -286,6 +268,11 @@ nvidia/Llama-3_3-Nemotron-Super-49B-v1: - quant_algo: FP8 kv_cache_quant_algo: FP8 accuracy: 92.42 +nvidia/Nemotron-H-8B-Base-8K: + - accuracy: 46.20 + - quant_algo: FP8 + kv_cache_quant_algo: FP8 + accuracy: 85.78 nvidia/Nemotron-MOE: - accuracy: 88.249 - quant_algo: FP8 diff --git a/tests/integration/defs/accuracy/references/mmlu.yaml b/tests/integration/defs/accuracy/references/mmlu.yaml index f4f7e0e98c65..ceeb5abd9f30 100644 --- a/tests/integration/defs/accuracy/references/mmlu.yaml +++ b/tests/integration/defs/accuracy/references/mmlu.yaml @@ -1,8 +1,3 @@ -meta-llama/Meta-Llama-3-8B-Instruct: - - accuracy: 67.74 - - quant_algo: NVFP4 - kv_cache_quant_algo: FP8 - accuracy: 63.47 meta-llama/Llama-3.1-8B: - accuracy: 66.06 - quant_algo: NVFP4 @@ -35,59 +30,6 @@ meta-llama/Llama-3.1-8B-Instruct: - quant_algo: NVFP4 kv_cache_quant_algo: FP8 accuracy: 65.11 -meta-llama/Llama-3.2-1B: - - quant_algo: W8A8_SQ_PER_CHANNEL_PER_TOKEN_PLUGIN - accuracy: 32.72 - - quant_algo: W8A8_SQ_PER_CHANNEL - accuracy: 32.07 - - quant_algo: W4A16_AWQ - accuracy: 30.56 - - quant_algo: W4A16_AWQ - kv_cache_quant_algo: INT8 - accuracy: 31.29 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 31.02 - - quant_algo: FP8_PER_CHANNEL_PER_TOKEN - accuracy: 33.97 - - quant_algo: FP8_PER_CHANNEL_PER_TOKEN - extra_acc_spec: meta_recipe - accuracy: 33.87 - - extra_acc_spec: max_attention_window_size=960 - accuracy: 32.82 -meta-llama/Llama-3.2-3B: - - accuracy: 57.92 - - spec_dec_algo: Eagle3 - accuracy: 57.92 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 60.60 -meta-llama/Llama-3.3-70B-Instruct: - - accuracy: 81.31 - - spec_dec_algo: Eagle3 - accuracy: 81.31 - - quant_algo: FP8 - spec_dec_algo: Eagle - accuracy: 81.31 - - quant_algo: FP8 - spec_dec_algo: Eagle3 - accuracy: 81.31 - - quant_algo: NVFP4 - kv_cache_quant_algo: FP8 - accuracy: 78.78 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 80.40 - - quant_algo: FP8 - accuracy: 80.40 -meta-llama/Llama-4-Scout-17B-16E-Instruct: - - accuracy: 80.00 - - quant_algo: NVFP4 - kv_cache_quant_algo: FP8 - accuracy: 79.60 - - quant_algo: FP8 - kv_cache_quant_algo: FP8 - accuracy: 78.58 mistralai/Mistral-7B-v0.1: - accuracy: 66 mistralai/Mistral-7B-Instruct-v0.3: @@ -309,6 +251,11 @@ nvidia/Llama-3.1-Nemotron-Nano-8B-v1: - quant_algo: FP8 kv_cache_quant_algo: FP8 accuracy: 57.12 +nvidia/Nemotron-H-8B-Base-8K: + - accuracy: 69.590 + - quant_algo: FP8 + kv_cache_quant_algo: FP8 + accuracy: 69.180 microsoft/Phi-4-mini-instruct: - accuracy: 68.98 - quant_algo: FP8 diff --git a/tests/integration/defs/accuracy/references/passkey_retrieval_128k.yaml b/tests/integration/defs/accuracy/references/passkey_retrieval_128k.yaml index f9f52a3ee867..0967ef424bce 100644 --- a/tests/integration/defs/accuracy/references/passkey_retrieval_128k.yaml +++ b/tests/integration/defs/accuracy/references/passkey_retrieval_128k.yaml @@ -1,2 +1 @@ -gradientai/Llama-3-8B-Instruct-Gradient-1048k: - - accuracy: 99 +{} diff --git a/tests/integration/defs/accuracy/test_llm_api_pytorch.py b/tests/integration/defs/accuracy/test_llm_api_pytorch.py index 322bfaf3d4d8..55bd9567aaf4 100644 --- a/tests/integration/defs/accuracy/test_llm_api_pytorch.py +++ b/tests/integration/defs/accuracy/test_llm_api_pytorch.py @@ -1037,166 +1037,6 @@ def test_fp8_beam_search(self, enable_cuda_graph, enable_padding, extra_acc_spec="beam_width=2") -class TestLlama3_2_3B(LlmapiAccuracyTestHarness): - MODEL_NAME = "meta-llama/Llama-3.2-3B" - MODEL_PATH = f"{llm_models_root()}/llama-3.2-models/Llama-3.2-3B" - EXAMPLE_FOLDER = "models/core/llama" - - def test_auto_dtype(self): - with LLM(self.MODEL_PATH) as llm: - task = CnnDailymail(self.MODEL_NAME) - task.evaluate(llm) - task = MMLU(self.MODEL_NAME) - task.evaluate(llm) - - @skip_pre_hopper - def test_fp8_prequantized(self): - model_path = f"{llm_models_root()}/llama-3.2-models/Llama-3.2-3B-Instruct-FP8" - with LLM(model_path) as llm: - assert llm.args.quant_config.quant_algo == QuantAlgo.FP8 - task = CnnDailymail(self.MODEL_NAME) - task.evaluate(llm) - task = MMLU(self.MODEL_NAME) - task.evaluate(llm) - - -@pytest.mark.timeout(7200) -@pytest.mark.skip_less_device_memory(80000) -class TestLlama3_3_70BInstruct(LlmapiAccuracyTestHarness): - MODEL_NAME = "meta-llama/Llama-3.3-70B-Instruct" - - @pytest.mark.skip_less_mpi_world_size(8) - def test_auto_dtype_tp8(self): - model_path = f"{llm_models_root()}/llama-3.3-models/Llama-3.3-70B-Instruct" - with LLM(model_path, tensor_parallel_size=8) as llm: - task = MMLU(self.MODEL_NAME) - task.evaluate(llm) - task = GSM8K(self.MODEL_NAME) - task.evaluate(llm) - task = GPQADiamond(self.MODEL_NAME) - task.evaluate(llm, - extra_evaluator_kwargs=dict(apply_chat_template=True)) - - @pytest.mark.skip_less_mpi_world_size(2) - def test_auto_dtype_tp2(self): - _run_multinode_accuracy( - f"{llm_models_root()}/llama-3.3-models/Llama-3.3-70B-Instruct", - self.MODEL_NAME, - benchmarks=["mmlu"]) - - @skip_pre_hopper - @pytest.mark.skip_less_mpi_world_size(8) - @parametrize_with_ids("torch_compile", [False, True]) - @parametrize_with_ids("eagle3_one_model", [True, False]) - def test_fp8_eagle3_tp8(self, eagle3_one_model, torch_compile): - model_path = f"{llm_models_root()}/modelopt-hf-model-hub/Llama-3.3-70B-Instruct-fp8" - eagle_model_dir = f"{llm_models_root()}/EAGLE3-LLaMA3.3-Instruct-70B" - kv_cache_config = KvCacheConfig(free_gpu_memory_fraction=0.6) - spec_config = Eagle3DecodingConfig(max_draft_len=3, - speculative_model=eagle_model_dir, - eagle3_one_model=eagle3_one_model) - torch_compile_config = _get_default_torch_compile_config(torch_compile) - pytorch_config = dict( - disable_overlap_scheduler=not eagle3_one_model, - cuda_graph_config=CudaGraphConfig(max_batch_size=1), - torch_compile_config=torch_compile_config) - with LLM(model_path, - max_batch_size=16, - tensor_parallel_size=8, - speculative_config=spec_config, - kv_cache_config=kv_cache_config, - **pytorch_config) as llm: - task = CnnDailymail(self.MODEL_NAME) - task.evaluate(llm) - - @pytest.mark.skip_less_device(4) - @skip_pre_hopper - @parametrize_with_ids("torch_compile", [False, True]) - def test_fp8_tp4(self, torch_compile): - model_path = f"{llm_models_root()}/llama-3.3-models/Llama-3.3-70B-Instruct-FP8" - kv_cache_config = KvCacheConfig(free_gpu_memory_fraction=0.5) - torch_compile_config = _get_default_torch_compile_config(torch_compile) - with LLM(model_path, - tensor_parallel_size=4, - max_seq_len=8192, - max_batch_size=32, - kv_cache_config=kv_cache_config, - torch_compile_config=torch_compile_config) as llm: - assert llm.args.quant_config.quant_algo == QuantAlgo.FP8 - sampling_params = SamplingParams( - max_tokens=256, - temperature=0.0, - add_special_tokens=False, - ) - task = MMLU(self.MODEL_NAME) - task.evaluate(llm, sampling_params=sampling_params) - task = GSM8K(self.MODEL_NAME) - task.evaluate(llm, sampling_params=sampling_params) - task = GPQADiamond(self.MODEL_NAME) - task.evaluate(llm, - extra_evaluator_kwargs=dict(apply_chat_template=True)) - - @pytest.mark.skip_less_device(4) - @skip_pre_blackwell - @parametrize_with_ids("torch_compile", [False, True]) - def test_nvfp4_tp4(self, torch_compile): - model_path = f"{llm_models_root()}/llama-3.3-models/Llama-3.3-70B-Instruct-FP4" - kv_cache_config = KvCacheConfig(free_gpu_memory_fraction=0.5) - torch_compile_config = _get_default_torch_compile_config(torch_compile) - with LLM(model_path, - tensor_parallel_size=4, - max_batch_size=32, - kv_cache_config=kv_cache_config, - torch_compile_config=torch_compile_config) as llm: - assert llm.args.quant_config.quant_algo == QuantAlgo.NVFP4 - sampling_params = SamplingParams( - max_tokens=256, - temperature=0.0, - add_special_tokens=False, - ) - task = MMLU(self.MODEL_NAME) - task.evaluate(llm, sampling_params=sampling_params) - task = GSM8K(self.MODEL_NAME) - task.evaluate(llm, sampling_params=sampling_params) - task = GPQADiamond(self.MODEL_NAME) - task.evaluate(llm, - extra_evaluator_kwargs=dict(apply_chat_template=True)) - - @pytest.mark.skip_less_device(4) - @skip_pre_blackwell - @parametrize_with_ids("enable_gemm_allreduce_fusion", [False, True]) - @parametrize_with_ids("torch_compile", [False, True]) - def test_fp4_tp2pp2(self, enable_gemm_allreduce_fusion, torch_compile): - model_path = f"{llm_models_root()}/llama-3.3-models/Llama-3.3-70B-Instruct-FP4" - kv_cache_config = KvCacheConfig(free_gpu_memory_fraction=0.5) - torch_compile_config = _get_default_torch_compile_config(torch_compile) - - with (mock.patch.dict( - os.environ, { - "TRTLLM_GEMM_ALLREDUCE_FUSION_ENABLED": - str(int(enable_gemm_allreduce_fusion)) - }), - LLM(model_path, - tensor_parallel_size=2, - pipeline_parallel_size=2, - max_batch_size=32, - kv_cache_config=kv_cache_config, - torch_compile_config=torch_compile_config) as llm): - assert llm.args.quant_config.quant_algo == QuantAlgo.NVFP4 - sampling_params = SamplingParams( - max_tokens=256, - temperature=0.0, - add_special_tokens=False, - ) - task = MMLU(self.MODEL_NAME) - task.evaluate(llm, sampling_params=sampling_params) - task = GSM8K(self.MODEL_NAME) - task.evaluate(llm, sampling_params=sampling_params) - task = GPQADiamond(self.MODEL_NAME) - task.evaluate(llm, - extra_evaluator_kwargs=dict(apply_chat_template=True)) - - class TestMinistral8BInstruct(LlmapiAccuracyTestHarness): MODEL_NAME = "mistralai/Ministral-8B-Instruct-2410" MODEL_PATH = f"{llm_models_root()}/Ministral-8B-Instruct-2410" diff --git a/tests/integration/defs/accuracy/test_llm_api_pytorch_encode.py b/tests/integration/defs/accuracy/test_llm_api_pytorch_encode.py index 1a3b7a8577ed..a92545d9218e 100644 --- a/tests/integration/defs/accuracy/test_llm_api_pytorch_encode.py +++ b/tests/integration/defs/accuracy/test_llm_api_pytorch_encode.py @@ -357,12 +357,6 @@ def test_qwen3_text_embedding_matches_huggingface(self, model_name, model_path): # Qwen2ForCausalLM — Qwen2-7B (distinct GQA head config, SwiGLU variant) # Qwen3ForCausalLM — Qwen3-0.6B (QKNorm, architecturally distinct from Qwen2) DECODER_MODELS = [ - # -- LlamaForCausalLM (covers Llama + Mistral family) -- - pytest.param( - "TinyLlama/TinyLlama-1.1B-Chat-v1.0", - f"{llm_models_root()}/llama-models-v2/TinyLlama-1.1B-Chat-v1.0", - id="tinyllama-1.1b", - ), # -- Gemma3ForCausalLM -- pytest.param( "google/gemma-3-1b-it", diff --git a/tests/integration/defs/conftest.py b/tests/integration/defs/conftest.py index c63d15ff0759..2426aaae39f1 100644 --- a/tests/integration/defs/conftest.py +++ b/tests/integration/defs/conftest.py @@ -568,8 +568,6 @@ def multimodal_model_root(request, llm_venv): if "neva-22b" in tllm_model_name: models_root = os.path.join(llm_models_root(), "neva") tllm_model_name = tllm_model_name + ".nemo" - elif "Llama-3.2" in tllm_model_name: - models_root = os.path.join(llm_models_root(), "llama-3.2-models") elif "Mistral-Small" in tllm_model_name: models_root = llm_models_root() @@ -686,40 +684,11 @@ def llm_gpt2b_lora_model_root(request): return ",".join(model_root_list) -@pytest.fixture(scope="module") -def llama_v2_tokenizer_model_root(): - models_root = llm_models_root() - assert models_root, "Did you set LLM_MODELS_ROOT?" - llama_v2_tokenizer_model_root = os.path.join(models_root, "llama-models-v2") - - assert os.path.exists( - llama_v2_tokenizer_model_root - ), f"{llama_v2_tokenizer_model_root} does not exist under NFS LLM_MODELS_ROOT dir" - return llama_v2_tokenizer_model_root - - @pytest.fixture(scope="function") def llama_model_root(request): models_root = llm_models_root() assert models_root, "Did you set LLM_MODELS_ROOT?" - if request.param == "llama-30b": - llama_model_root = os.path.join(models_root, "llama-models", - "llama-30b-hf") - elif request.param == "TinyLlama-1.1B-Chat-v1.0": - llama_model_root = os.path.join(models_root, "llama-models-v2", - "TinyLlama-1.1B-Chat-v1.0") - elif request.param == "llama-v3-8b-hf": - llama_model_root = os.path.join(models_root, "llama-models-v3", "8B") - elif request.param == "llama-v3-8b-instruct-hf": - llama_model_root = os.path.join(models_root, "llama-models-v3", - "llama-v3-8b-instruct-hf") - elif request.param == "Llama-3-8B-Instruct-Gradient-1048k": - llama_model_root = os.path.join(models_root, "llama-models-v3", - "Llama-3-8B-Instruct-Gradient-1048k") - elif request.param == "Llama-3-70B-Instruct-Gradient-1048k": - llama_model_root = os.path.join(models_root, "llama-models-v3", - "Llama-3-70B-Instruct-Gradient-1048k") - elif request.param == "llama-3.1-8b": + if request.param == "llama-3.1-8b": llama_model_root = os.path.join(models_root, "llama-3.1-model", "Meta-Llama-3.1-8B") elif request.param == "llama-3.1-8b-instruct-hf-fp8": @@ -731,47 +700,14 @@ def llama_model_root(request): elif request.param == "llama-3.1-8b-hf-nvfp4": llama_model_root = os.path.join(models_root, "nvfp4-quantized", "Meta-Llama-3.1-8B") - elif request.param == "llama-3.2-1b": - llama_model_root = os.path.join(models_root, "llama-3.2-models", - "Llama-3.2-1B") - elif request.param == "llama-3.2-1b-instruct": - llama_model_root = os.path.join(models_root, "llama-3.2-models", - "Llama-3.2-1B-Instruct") - elif request.param == "llama-3.2-3b": - llama_model_root = os.path.join(models_root, "llama-3.2-models", - "Llama-3.2-3B") - elif request.param == "llama-3.2-3b-instruct": - llama_model_root = os.path.join(models_root, "llama-3.2-models", - "Llama-3.2-3B-Instruct") - elif request.param == "llama-3.3-70b-instruct": - llama_model_root = os.path.join(models_root, "llama-3.3-models", - "Llama-3.3-70B-Instruct") + elif request.param == "Qwen3-0.6B": + llama_model_root = os.path.join(models_root, "Qwen3", "Qwen3-0.6B") assert os.path.exists( llama_model_root ), f"{llama_model_root} does not exist under NFS LLM_MODELS_ROOT dir" return llama_model_root -@pytest.fixture(scope="function") -def code_llama_model_root(request): - "get CodeLlama model data" - models_root = llm_models_root() - assert models_root, "Did you set LLM_MODELS_ROOT?" - if request.param == "CodeLlama-7b-Instruct": - codellama_model_root = os.path.join(models_root, "codellama", - "CodeLlama-7b-Instruct-hf") - elif request.param == "CodeLlama-13b-Instruct": - codellama_model_root = os.path.join(models_root, "codellama", - "CodeLlama-13b-Instruct-hf") - elif request.param == "CodeLlama-34b-Instruct": - codellama_model_root = os.path.join(models_root, "codellama", - "CodeLlama-34b-Instruct-hf") - elif request.param == "CodeLlama-70b-hf": - codellama_model_root = os.path.join(models_root, "codellama", - "CodeLlama-70b-hf") - return codellama_model_root - - @pytest.fixture(scope="function") def draft_target_model_roots(request): models_root = llm_models_root() diff --git a/tests/integration/defs/disaggregated/test_ad_disagg.py b/tests/integration/defs/disaggregated/test_ad_disagg.py index cbd743f6a499..7b120311e153 100644 --- a/tests/integration/defs/disaggregated/test_ad_disagg.py +++ b/tests/integration/defs/disaggregated/test_ad_disagg.py @@ -64,7 +64,7 @@ def skip_b300(): "OMPI_UNIVERSE_SIZE", ) AUTODEPLOY_DISAGG_SEED = 1234 -REDUCED_TINYLLAMA_LAYERS = 2 +REDUCED_QWEN3_LAYERS = 2 REDUCED_DEEPSEEK_LAYERS = 2 LLAMA_EAGLE3_EXPECTED_TEXT = " Berlin\nWhat is the capital of France? Paris\nWhat is the capital of" LLAMA_EAGLE3_EXPECTED_TOKEN_IDS = [ @@ -90,7 +90,7 @@ def skip_b300(): MODEL_PATHS = { "EAGLE3-LLaMA3.1-Instruct-8B": "EAGLE3-LLaMA3.1-Instruct-8B", "Llama-3.1-8B-Instruct": "llama-3.1-model/Llama-3.1-8B-Instruct/", - "TinyLlama-1.1B-Chat-v1.0": "llama-models-v2/TinyLlama-1.1B-Chat-v1.0", + "Qwen3-0.6B": "Qwen3/Qwen3-0.6B", "DeepSeek-V3-Lite": "DeepSeek-V3-Lite/bf16", } @@ -279,9 +279,9 @@ def run_aggregate_generation( # --------------------------------------------------------------------------- -def reduced_tinyllama_config(extra_config=None): +def reduced_qwen3_config(extra_config=None): config = { - "model_kwargs": {"num_hidden_layers": REDUCED_TINYLLAMA_LAYERS}, + "model_kwargs": {"num_hidden_layers": REDUCED_QWEN3_LAYERS}, "max_batch_size": 4, "max_seq_len": 512, "max_num_tokens": 256, @@ -458,7 +458,7 @@ def reduced_model_config(model, extra_config=None): if "DeepSeek-V3-Lite" in model: config = reduced_deepseek_v3_mla_config() else: - config = reduced_tinyllama_config() + config = reduced_qwen3_config() if extra_config: config.update(extra_config) return config @@ -467,8 +467,8 @@ def reduced_model_config(model, extra_config=None): def reduced_model_cases(): return [ pytest.param( - "TinyLlama-1.1B-Chat-v1.0", - id="tinyllama", + "Qwen3-0.6B", + id="qwen3_0.6b", ), pytest.param( "DeepSeek-V3-Lite", @@ -562,7 +562,7 @@ def test_disaggregated_logits(model): # The MLA generation worker reconstructs logits from the compressed KV latent # through a different kernel/batching path than the single aggregate pass, so # bf16 rounding yields ~1-ULP logit differences. Use a looser tolerance for the - # MLA (DeepSeek) case; MHA (tinyllama) stays tight. The functional checks above + # MLA (DeepSeek) case; MHA (Qwen3-0.6B) stays tight. The functional checks above # (text/token_ids equality) remain strict for both. if "DeepSeek-V3-Lite" in model: rtol, atol = 1e-1, 1e-1 @@ -576,33 +576,6 @@ def test_disaggregated_logits(model): ) -@pytest.mark.skip_less_device_memory(30000) -@pytest.mark.timeout(600) -def test_tinyllama_batch_handoff_semantic_slots(): - prompts = capital_completion_prompts() - expected_capitals = ["Berlin", "Paris", "Rome", "Madrid"] - sampling_params_kwargs = { - "max_tokens": 12, - "ignore_eos": True, - "top_k": 1, - "seed": AUTODEPLOY_DISAGG_SEED, - } - outputs = run_sequential_batch_handoff( - "TinyLlama-1.1B-Chat-v1.0", - generation_overlap=True, - prompts=prompts, - sampling_params_kwargs=sampling_params_kwargs, - ) - - for expected_capital, context_output, generation_output in zip( - expected_capitals, outputs["context"], outputs["generation"], strict=True - ): - assert_context_handoff_metadata(context_output) - assert expected_capital.lower() in generation_output.text.lower(), response_summary( - outputs["generation"] - ) - - @pytest.mark.parametrize( "model", reduced_model_cases(), @@ -942,13 +915,13 @@ def run_context_then_generation_handoff( @pytest.mark.timeout(600) def test_async_generation_matches_aggregate(): aggregate_output = run_aggregate_generation( - "TinyLlama-1.1B-Chat-v1.0", + "Qwen3-0.6B", world_size=1, prompt="What is the capital of Germany?", sampling_params_kwargs={"max_tokens": 10, "ignore_eos": True}, ) outputs = run_context_then_generation_handoff( - "TinyLlama-1.1B-Chat-v1.0", + "Qwen3-0.6B", worker_world_sizes=(1, 1), generation_overlap=True, prompt="What is the capital of Germany?", @@ -980,13 +953,13 @@ def test_async_generation_no_overlap_matches_aggregate(): """ sampling_params_kwargs = {"max_tokens": 10, "ignore_eos": True} aggregate_output = run_aggregate_generation( - "TinyLlama-1.1B-Chat-v1.0", + "Qwen3-0.6B", world_size=1, prompt="What is the capital of Germany?", sampling_params_kwargs=sampling_params_kwargs, ) outputs = run_context_then_generation_handoff( - "TinyLlama-1.1B-Chat-v1.0", + "Qwen3-0.6B", worker_world_sizes=(1, 1), generation_overlap=False, prompt="What is the capital of Germany?", @@ -1005,13 +978,13 @@ def test_async_generation_no_overlap_matches_aggregate(): @pytest.mark.timeout(900) def test_async_sharded_generation_handoff(): aggregate_output = run_aggregate_generation( - "TinyLlama-1.1B-Chat-v1.0", + "Qwen3-0.6B", world_size=2, prompt="What is the capital of Germany?", sampling_params_kwargs={"max_tokens": 10, "ignore_eos": True}, ) outputs = run_context_then_generation_handoff( - "TinyLlama-1.1B-Chat-v1.0", + "Qwen3-0.6B", worker_world_sizes=(2, 2), generation_overlap=True, prompt="What is the capital of Germany?", diff --git a/tests/integration/defs/disaggregated/test_ad_disagg_trtllm_serve.py b/tests/integration/defs/disaggregated/test_ad_disagg_trtllm_serve.py index 9698e6799ef9..468881a9b357 100644 --- a/tests/integration/defs/disaggregated/test_ad_disagg_trtllm_serve.py +++ b/tests/integration/defs/disaggregated/test_ad_disagg_trtllm_serve.py @@ -15,22 +15,11 @@ import asyncio import os -from pathlib import Path import pytest import requests -from defs.common import get_free_port_in_ci as get_free_port -from defs.conftest import check_device_contain, llm_models_root -from disagg_test_utils import ( - CHECK_STATUS_INTERVAL, - HEARTBEAT_INTERVAL, - INACTIVE_TIMEOUT, - run_ctx_worker, - run_disagg_server, - run_gen_worker, - terminate, -) -from openai import OpenAI +from defs.conftest import check_device_contain +from disagg_test_utils import CHECK_STATUS_INTERVAL, HEARTBEAT_INTERVAL, INACTIVE_TIMEOUT pytest_plugins = ["disagg_test_utils"] @@ -48,15 +37,10 @@ def skip_b300(): SERVER_READY_REQUEST_TIMEOUT_S = 5 OPENAI_REQUEST_TIMEOUT_S = 60 PROXY_PORT_MAX_RETRIES = 5 -TINYLLAMA_MODEL_DIR = "llama-models-v2/TinyLlama-1.1B-Chat-v1.0" AUTODEPLOY_BACKEND = "_autodeploy" EXPECTED_COMPLETION_SUBSTRING = "Berlin" -def tinyllama_model_path(): - return str(Path(llm_models_root()) / TINYLLAMA_MODEL_DIR) - - def worker_cuda_devices(num_workers): visible_devices = os.environ.get("CUDA_VISIBLE_DEVICES") if visible_devices: @@ -161,106 +145,3 @@ async def wait_for_disagg_server_ready_or_exit(port, processes, timeout, request f"Timed out after {timeout}s waiting for disaggregated server on port {port}; " f"{last_readiness_error}" ) - - -@pytest.mark.skip_less_device_memory(30000) -@pytest.mark.skip_less_device(2) -@pytest.mark.timeout(900) -@pytest.mark.asyncio(loop_scope="module") -async def test_openai_completion(work_dir): - """Smoke test AutoDeploy disagg through trtllm-serve and the OpenAI API. - - The lower-level tests in ``test_ad_disagg.py`` drive AutoDeploy workers - directly and inspect context/generation handoff metadata. This test instead - verifies the trtllm-serve deployment shape: context worker, generation - worker, disaggregated proxy, and an OpenAI-compatible completion request. - """ - model = tinyllama_model_path() - ctx_device, gen_device = worker_cuda_devices(2) - - last_port_conflict = None - response = None - for attempt in range(PROXY_PORT_MAX_RETRIES): - disagg_port = get_free_port() - disagg_cluster = disagg_cluster_config(disagg_port) - ctx_worker = None - gen_worker = None - disagg_server = None - - try: - # Use the same service-discovery path as the broader PyTorch disagg - # tests for worker ports. Passing port=0 lets each trtllm-serve worker - # bind an OS-selected port in the child process and register that port - # with the disaggregated proxy. - ctx_worker = run_ctx_worker( - model, - autodeploy_worker_config(disagg_cluster, disable_overlap_scheduler=True), - work_dir, - port=0, - device=ctx_device, - ) - gen_worker = run_gen_worker( - model, - autodeploy_worker_config(disagg_cluster), - work_dir, - port=0, - device=gen_device, - ) - disagg_server = run_disagg_server( - proxy_config(disagg_port, disagg_cluster), - work_dir, - disagg_port, - save_log=True, - ) - try: - await wait_for_disagg_server_ready_or_exit( - disagg_port, - { - "context worker": ctx_worker, - "generation worker": gen_worker, - "disaggregated proxy": disagg_server, - }, - SERVER_START_TIMEOUT_S, - SERVER_READY_REQUEST_TIMEOUT_S, - ) - except RuntimeError as exc: - last_port_conflict = exc - if "disaggregated proxy" not in str(exc) or ( - "EADDRINUSE" not in str(exc) - and "address already in use" not in str(exc).lower() - ): - raise - print( - f"AutoDeploy disagg serve attempt {attempt + 1} of {PROXY_PORT_MAX_RETRIES} " - f"failed with proxy port conflict, retrying: {exc}" - ) - continue - - client = OpenAI( - api_key="tensorrt_llm", - base_url=f"http://localhost:{disagg_port}/v1", - timeout=OPENAI_REQUEST_TIMEOUT_S, - max_retries=0, - ) - response = client.completions.create( - model=model, - prompt="What is the capital of Germany?", - max_tokens=32, - temperature=0, - extra_body={"ignore_eos": True}, - ) - break - finally: - terminate(ctx_worker, gen_worker, disagg_server) - - if response is None: - raise RuntimeError( - f"Failed to start AutoDeploy disagg serve smoke after {PROXY_PORT_MAX_RETRIES} " - "proxy port attempts" - ) from last_port_conflict - - assert response.choices - response_text = response.choices[0].text - assert EXPECTED_COMPLETION_SUBSTRING in response_text, ( - f"expected {EXPECTED_COMPLETION_SUBSTRING!r} in response, got {response_text!r}" - ) diff --git a/tests/integration/defs/disaggregated/test_auto_scaling.py b/tests/integration/defs/disaggregated/test_auto_scaling.py index efa0570e44f9..c5d76037f8a2 100644 --- a/tests/integration/defs/disaggregated/test_auto_scaling.py +++ b/tests/integration/defs/disaggregated/test_auto_scaling.py @@ -42,8 +42,7 @@ def worker_env(): @pytest.fixture def model_name(): - model_path = os.path.join(llm_models_root(), - "llama-models-v2/TinyLlama-1.1B-Chat-v1.0") + model_path = os.path.join(llm_models_root(), "Qwen3/Qwen3-0.6B") assert os.path.exists(model_path), f"Model path {model_path} does not exist" return model_path diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config.yaml index a29c2a5303f8..8c5c7ff1222f 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config.yaml @@ -1,5 +1,5 @@ hostname: localhost -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B free_gpu_memory_fraction: 0.25 backend: pytorch disable_overlap_scheduler: true diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_cache_aware_balance.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_cache_aware_balance.yaml index a9bf2587d23e..383e750a3a11 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_cache_aware_balance.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_cache_aware_balance.yaml @@ -1,4 +1,4 @@ -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B hostname: localhost backend: pytorch cuda_graph_config: null diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_cache_reuse.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_cache_reuse.yaml index e7b371a6479e..49415ed6cafc 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_cache_reuse.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_cache_reuse.yaml @@ -1,5 +1,5 @@ hostname: localhost -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B backend: pytorch cuda_graph_config: null disable_overlap_scheduler: true diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_conditional.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_conditional.yaml index 26aaeac42d90..4b523838979b 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_conditional.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_conditional.yaml @@ -1,4 +1,4 @@ -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B hostname: localhost backend: pytorch cuda_graph_config: null diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_conversation.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_conversation.yaml index fbbfd0d21e1d..23d19a8bac86 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_conversation.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_conversation.yaml @@ -1,5 +1,5 @@ hostname: localhost -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B backend: pytorch cuda_graph_config: null disable_overlap_scheduler: true diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_conversation_workers.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_conversation_workers.yaml index e894d036def2..6fa0b1b7c019 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_conversation_workers.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_conversation_workers.yaml @@ -1,4 +1,4 @@ -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B hostname: localhost backend: pytorch cuda_graph_config: null diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxpp2_genpp2.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxpp2_genpp2.yaml index c04b34238c6b..67264c5d0eac 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxpp2_genpp2.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxpp2_genpp2.yaml @@ -1,4 +1,4 @@ -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B hostname: localhost backend: pytorch cuda_graph_config: null diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxpp2_gentp2.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxpp2_gentp2.yaml index 76e44e23a12d..9a454405b9c0 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxpp2_gentp2.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxpp2_gentp2.yaml @@ -1,4 +1,4 @@ -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B hostname: localhost backend: pytorch cuda_graph_config: null diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxpp4_genpp4.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxpp4_genpp4.yaml index ffee6430abcc..fff4d54cde09 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxpp4_genpp4.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxpp4_genpp4.yaml @@ -1,4 +1,4 @@ -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B hostname: localhost backend: pytorch cuda_graph_config: null diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxpp4_gentp4.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxpp4_gentp4.yaml index c176aa863b61..746539b9aaa9 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxpp4_gentp4.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxpp4_gentp4.yaml @@ -1,4 +1,4 @@ -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B hostname: localhost backend: pytorch cuda_graph_config: null diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_genpp2.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_genpp2.yaml index 8d6821cd996c..e59112295147 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_genpp2.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_genpp2.yaml @@ -1,4 +1,4 @@ -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B hostname: localhost backend: pytorch cuda_graph_config: null diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_gentp1.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_gentp1.yaml index 840ba25e021d..d4d8d864b453 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_gentp1.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_gentp1.yaml @@ -1,5 +1,5 @@ hostname: localhost -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B free_gpu_memory_fraction: 0.25 backend: pytorch cuda_graph_config: null diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2pp2_gentp2pp2.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2pp2_gentp2pp2.yaml index d80795b727ac..91e3f3adde7d 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2pp2_gentp2pp2.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2pp2_gentp2pp2.yaml @@ -1,4 +1,4 @@ -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B hostname: localhost backend: pytorch cuda_graph_config: null diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_cuda_graph_padding.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_cuda_graph_padding.yaml index 1f9e42d73237..6cda7d0a77a6 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_cuda_graph_padding.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_cuda_graph_padding.yaml @@ -1,4 +1,4 @@ -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B hostname: localhost backend: pytorch context_servers: diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_diff_max_tokens.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_diff_max_tokens.yaml index c07260248822..98fee7f6394f 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_diff_max_tokens.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_diff_max_tokens.yaml @@ -1,5 +1,5 @@ hostname: localhost -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B free_gpu_memory_fraction: 0.25 backend: pytorch cuda_graph_config: null diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_gen_only.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_gen_only.yaml index 9253f421cfcd..a2dbb3f88d83 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_gen_only.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_gen_only.yaml @@ -1,5 +1,5 @@ hostname: localhost -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B backend: pytorch cuda_graph_config: null context_servers: diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_gen_only_bs1.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_gen_only_bs1.yaml index 67494b24ff0b..e18d6c0e00b8 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_gen_only_bs1.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_gen_only_bs1.yaml @@ -1,4 +1,4 @@ -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B hostname: localhost backend: pytorch cuda_graph_config: null diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_gen_only_insufficient_kv.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_gen_only_insufficient_kv.yaml index 9f65a7908e36..a06c29b74754 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_gen_only_insufficient_kv.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_gen_only_insufficient_kv.yaml @@ -1,5 +1,5 @@ hostname: localhost -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B backend: pytorch cuda_graph_config: null context_servers: diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_gen_only_kv_cache_aware.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_gen_only_kv_cache_aware.yaml index 4e40bbf006a1..b107f8b9027c 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_gen_only_kv_cache_aware.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_gen_only_kv_cache_aware.yaml @@ -1,5 +1,5 @@ hostname: localhost -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B backend: pytorch cuda_graph_config: null disable_overlap_scheduler: true diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_llama4_kv_cache_overflow.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_llama4_kv_cache_overflow.yaml deleted file mode 100644 index fa65a710981c..000000000000 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_llama4_kv_cache_overflow.yaml +++ /dev/null @@ -1,45 +0,0 @@ -model: llama4-models/nvidia/Llama-4-Maverick-17B-128E-Instruct-FP8 -hostname: localhost -backend: pytorch -context_servers: - num_instances: 1 - tensor_parallel_size: 4 - pipeline_parallel_size: 1 - moe_expert_parallel_size: 1 - enable_attention_dp: false - max_num_tokens: 8192 - max_seq_len: 257000 - max_input_len: 256000 - max_batch_size: 1 - trust_remote_code: true - enable_chunked_prefill: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.3 - disable_overlap_scheduler: true - cuda_graph_config: null - cache_transceiver_config: - backend: UCX - # Intentionally small to reproduce buffer overflow bug - max_tokens_in_buffer: 2048 -generation_servers: - num_instances: 1 - tensor_parallel_size: 4 - pipeline_parallel_size: 1 - moe_expert_parallel_size: 1 - enable_attention_dp: false - max_num_tokens: 8192 - max_seq_len: 257000 - max_input_len: 256000 - max_batch_size: 1 - trust_remote_code: true - enable_chunked_prefill: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.3 - disable_overlap_scheduler: true - cuda_graph_config: null - cache_transceiver_config: - backend: UCX - # Intentionally small to reproduce buffer overflow bug - max_tokens_in_buffer: 2048 diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_load_balance.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_load_balance.yaml index 8540c6f555f6..6d6bede95630 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_load_balance.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_load_balance.yaml @@ -1,4 +1,4 @@ -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B hostname: localhost backend: pytorch cuda_graph_config: null diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_load_balancing.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_load_balancing.yaml index 144c9af0f72e..d0868f8277f5 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_load_balancing.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_load_balancing.yaml @@ -1,5 +1,5 @@ hostname: localhost -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B backend: pytorch cuda_graph_config: null disable_overlap_scheduler: true diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_metrics.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_metrics.yaml index 4bb52cc134f9..f4f7e0620e49 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_metrics.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_metrics.yaml @@ -1,5 +1,5 @@ hostname: localhost -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B free_gpu_memory_fraction: 0.25 backend: "pytorch" cuda_graph_config: null diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_mixed.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_mixed.yaml index cf7478ce8588..4582e152fb8f 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_mixed.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_mixed.yaml @@ -1,5 +1,5 @@ hostname: localhost -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B free_gpu_memory_fraction: 0.25 backend: "pytorch" cuda_graph_config: null diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_multi_orchestrator.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_multi_orchestrator.yaml index 970c2e276647..2382366ab5c3 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_multi_orchestrator.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_multi_orchestrator.yaml @@ -1,5 +1,5 @@ hostname: localhost -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B num_workers: 4 free_gpu_memory_fraction: 0.25 backend: pytorch diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_ngram.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_ngram.yaml index 4d0e7f804368..95f7278c2f9b 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_ngram.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_ngram.yaml @@ -1,5 +1,5 @@ hostname: localhost -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B free_gpu_memory_fraction: 0.1 backend: pytorch disable_overlap_scheduler: true diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_overlap.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_overlap.yaml index 391f95605b2c..4b796be5ed65 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_overlap.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_overlap.yaml @@ -1,4 +1,4 @@ -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B hostname: localhost backend: pytorch cuda_graph_config: null diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_overlap_gen_first.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_overlap_gen_first.yaml index 50e8f172101b..af67d32b0808 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_overlap_gen_first.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_overlap_gen_first.yaml @@ -1,4 +1,4 @@ -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B hostname: localhost port: 8000 backend: "pytorch" diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_overlap_gen_first_pp4.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_overlap_gen_first_pp4.yaml index 4f50e9b57150..70134e60b0cf 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_overlap_gen_first_pp4.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_overlap_gen_first_pp4.yaml @@ -1,4 +1,4 @@ -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B hostname: localhost port: 8000 backend: "pytorch" diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_overlap_transceiver_runtime_python.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_overlap_transceiver_runtime_python.yaml index 33b4d256ad54..b44337dc6996 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_overlap_transceiver_runtime_python.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_overlap_transceiver_runtime_python.yaml @@ -1,4 +1,4 @@ -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B hostname: localhost port: 8000 backend: "pytorch" diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_overlap_transceiver_runtime_python_bounce.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_overlap_transceiver_runtime_python_bounce.yaml index b4b3cd1234b9..c3939076edfd 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_overlap_transceiver_runtime_python_bounce.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_overlap_transceiver_runtime_python_bounce.yaml @@ -7,7 +7,7 @@ # TRTLLM_KV_CACHE_BOUNCE_MIN_BLOCKS env (set by the test) so the ordinary short test prompts still # take the coalesced-bounce WRITE path (the production default of 96 would need a ~2k-token prompt). # GB200/GB300 only, since the bounce arena is fabric (MNNVL) VMM memory. -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B hostname: localhost port: 8000 backend: "pytorch" diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_python_transceiver_host_offload.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_python_transceiver_host_offload.yaml index 7cfdef404168..85fc3519fef1 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_python_transceiver_host_offload.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_python_transceiver_host_offload.yaml @@ -1,4 +1,4 @@ -model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 +model: Qwen3/Qwen3-0.6B hostname: localhost port: 8000 backend: "pytorch" diff --git a/tests/integration/defs/disaggregated/test_disaggregated.py b/tests/integration/defs/disaggregated/test_disaggregated.py index 6789ee0c91a7..63c0c3c69ea2 100644 --- a/tests/integration/defs/disaggregated/test_disaggregated.py +++ b/tests/integration/defs/disaggregated/test_disaggregated.py @@ -384,8 +384,6 @@ def get_test_config(test_desc, example_dir, test_root): f"{test_configs_root}/disagg_config_ctxtp2ep2pp2_gentp4_deepseek_v3_lite_one_mtp_block_reuse_chunked.yaml", "deepseek_v3_lite_bf16_empty_batch": f"{test_configs_root}/disagg_config_deepseek_v3_lite_empty_batch.yaml", - "llama4_kv_cache_overflow": - f"{test_configs_root}/disagg_config_llama4_kv_cache_overflow.yaml", "deepseek_v3_lite_bf16_tllm_gen_helix": f"{test_configs_root}/disagg_config_ctxtp2_gentp1cp2_deepseek_v3_lite_bf16_tllm_gen.yaml", "deepseek_r1_v2_fp4_stress": @@ -1091,29 +1089,42 @@ def run_disaggregated_test(example_dir, shutil.rmtree(work_dir, ignore_errors=True) -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) -def test_disaggregated_diff_max_tokens(disaggregated_test_root, - disaggregated_example_root, llm_venv, - llama_model_root): - setup_model_symlink(llm_venv, llama_model_root, - "TinyLlama/TinyLlama-1.1B-Chat-v1.0") +# --------------------------------------------------------------------------- +# Qwen3-0.6B disaggregated tests +# +# These tests port the code-path coverage that was previously carried by +# Qwen3/Qwen3-0.6B. The model is resolved directly from +# llm_models_root() so no fixture parametrize is needed. +# --------------------------------------------------------------------------- + +_QWEN3_MODEL_SUBPATH = os.path.join("Qwen3", "Qwen3-0.6B") +_QWEN3_HF_ID = "Qwen3/Qwen3-0.6B" + + +def _qwen3_model_root() -> str: + """Return the NFS path to Qwen3-0.6B under LLM_MODELS_ROOT.""" + return os.path.join(llm_models_root(), _QWEN3_MODEL_SUBPATH) + + +@pytest.mark.skip_less_device(2) +@pytest.mark.parametrize("llama_model_root", ['Qwen3-0.6B'], indirect=True) +def test_disaggregated_multi_gpu(disaggregated_test_root, + disaggregated_example_root, llm_venv, + llama_model_root): + setup_model_symlink(llm_venv, llama_model_root, _QWEN3_HF_ID) run_disaggregated_test(disaggregated_example_root, - "2_ranks_diff_max_tokens", + "4_ranks", env=llm_venv._new_env, - prompt_file="long_prompts.json", model_path=llama_model_root, cwd=llm_venv.get_working_directory()) -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) +@pytest.mark.parametrize("llama_model_root", ['Qwen3-0.6B'], indirect=True) def test_disaggregated_single_gpu(disaggregated_test_root, disaggregated_example_root, llm_venv, llama_model_root): - setup_model_symlink(llm_venv, llama_model_root, - "TinyLlama/TinyLlama-1.1B-Chat-v1.0") + setup_model_symlink(llm_venv, llama_model_root, _QWEN3_HF_ID) env = llm_venv._new_env.copy() env["CUDA_VISIBLE_DEVICES"] = "0" @@ -1124,128 +1135,98 @@ def test_disaggregated_single_gpu(disaggregated_test_root, cwd=llm_venv.get_working_directory()) -def _verify_mamba_bs1_concurrency2(server_url: str) -> None: - - async def run() -> None: - timeout = aiohttp.ClientTimeout(total=120) - prompts = ( - "Write one sentence about disaggregated inference.", - "Write one sentence about recurrent-state transfer.", - ) +@pytest.mark.parametrize("llama_model_root", ['Qwen3-0.6B'], indirect=True) +def test_disaggregated_overlap(disaggregated_test_root, llm_venv, + disaggregated_example_root, llama_model_root): + setup_model_symlink(llm_venv, llama_model_root, _QWEN3_HF_ID) - async def send(session: aiohttp.ClientSession, prompt: str) -> None: - payload = { - "model": MAMBA_BS1_CONCURRENCY2_MODEL, - "prompt": prompt, - "max_tokens": 16, - "temperature": 0, - "ignore_eos": True, - } - async with session.post(f"{server_url}/v1/completions", - json=payload, - timeout=timeout) as response: - body = await response.json() - assert response.status == 200, body - assert body.get("choices"), body + def post_client_test(server_url: str): + verify_usage_with_cache_reuse(server_url, _QWEN3_HF_ID) - async with aiohttp.ClientSession() as session: - await asyncio.gather(*(send(session, prompt) for prompt in prompts)) + run_disaggregated_test(disaggregated_example_root, + "overlap", + env=llm_venv._new_env, + post_client_test=post_client_test, + model_path=llama_model_root, + cwd=llm_venv.get_working_directory()) - asyncio.run(run()) +@pytest.mark.parametrize("llama_model_root", ['Qwen3-0.6B'], indirect=True) +def test_disaggregated_cache_aware_balance(disaggregated_test_root, llm_venv, + disaggregated_example_root, + llama_model_root): + setup_model_symlink(llm_venv, llama_model_root, _QWEN3_HF_ID) -@skip_pre_blackwell -@pytest.mark.timeout(900) -def test_disaggregated_mamba_bs1_concurrency2(disaggregated_example_root, - llm_venv): - model_path = f"{llm_models_root()}/{MAMBA_BS1_CONCURRENCY2_MODEL}" - env = llm_venv._new_env.copy() - repo_root = os.path.abspath( - os.path.join(os.path.dirname(__file__), "../../../..")) - env["LLM_ROOT"] = repo_root - env["PYTHONPATH"] = os.pathsep.join(path for path in (repo_root, - env.get("PYTHONPATH")) - if path) - env.pop("TRTLLM_DISABLE_KV_CACHE_TRANSFER_OVERLAP", None) - env["TRTLLM_NIXL_NUM_THREADS"] = "1" - worker_env = {"TRTLLM_DISAGG_BENCHMARK_GEN_ONLY": "1"} - run_disaggregated_test( - disaggregated_example_root, - "mamba_bs1_concurrency2", - num_iters=0, - env=env, - model_path=model_path, - cwd=llm_venv.get_working_directory(), - post_client_test=_verify_mamba_bs1_concurrency2, - ctx_env=worker_env, - gen_env=worker_env, - share_gpu=True, - server_start_timeout=600, - ) + run_disaggregated_test(disaggregated_example_root, + "cache_aware_balance", + env=llm_venv._new_env, + model_path=llama_model_root, + cwd=llm_venv.get_working_directory()) -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) -def test_disaggregated_tinyllama_multi_orchestrator(disaggregated_test_root, - disaggregated_example_root, - llm_venv, llama_model_root): - setup_model_symlink(llm_venv, llama_model_root, - "TinyLlama/TinyLlama-1.1B-Chat-v1.0") +@pytest.mark.parametrize("llama_model_root", ['Qwen3-0.6B'], indirect=True) +def test_disaggregated_conditional(disaggregated_test_root, llm_venv, + disaggregated_example_root, + llama_model_root): + setup_model_symlink(llm_venv, llama_model_root, _QWEN3_HF_ID) - env = llm_venv._new_env.copy() - env["CUDA_VISIBLE_DEVICES"] = "0" run_disaggregated_test(disaggregated_example_root, - "multi_orchestrator", - num_iters=1, - env=env, + "conditional", + env=llm_venv._new_env, model_path=llama_model_root, cwd=llm_venv.get_working_directory()) -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) -def test_disaggregated_benchmark_gen_only(disaggregated_test_root, - disaggregated_example_root, llm_venv, - llama_model_root): - setup_model_symlink(llm_venv, llama_model_root, - "TinyLlama/TinyLlama-1.1B-Chat-v1.0") +@pytest.mark.parametrize("llama_model_root", ['Qwen3-0.6B'], indirect=True) +def test_disaggregated_chat_completion_tool_calls(disaggregated_test_root, + llm_venv, + disaggregated_example_root, + llama_model_root): + setup_model_symlink(llm_venv, llama_model_root, _QWEN3_HF_ID) - env = llm_venv._new_env.copy() - env['TRTLLM_DISAGG_BENCHMARK_GEN_ONLY'] = '1' run_disaggregated_test(disaggregated_example_root, - "gen_only", - env=env, + "tool_calls", + num_iters=1, + prompt_file="tool_call_prompts.json", + env=llm_venv._new_env, model_path=llama_model_root, cwd=llm_venv.get_working_directory()) -@pytest.mark.parametrize("router_type", - ["load_balancing", "kv_cache_aware", "conversation"]) -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) -def test_disaggregated_router(disaggregated_test_root, - disaggregated_example_root, llm_venv, - llama_model_root, router_type): - setup_model_symlink(llm_venv, llama_model_root, - "TinyLlama/TinyLlama-1.1B-Chat-v1.0") +def test_disaggregated_diff_max_tokens(disaggregated_test_root, + disaggregated_example_root, llm_venv): + qwen3 = _qwen3_model_root() + setup_model_symlink(llm_venv, qwen3, _QWEN3_HF_ID) run_disaggregated_test(disaggregated_example_root, - router_type, + "2_ranks_diff_max_tokens", env=llm_venv._new_env, - model_path=llama_model_root, + prompt_file="long_prompts.json", + model_path=qwen3, + cwd=llm_venv.get_working_directory()) + + +def test_disaggregated_benchmark_gen_only(disaggregated_test_root, + disaggregated_example_root, llm_venv): + qwen3 = _qwen3_model_root() + setup_model_symlink(llm_venv, qwen3, _QWEN3_HF_ID) + + env = llm_venv._new_env.copy() + env['TRTLLM_DISAGG_BENCHMARK_GEN_ONLY'] = '1' + run_disaggregated_test(disaggregated_example_root, + "gen_only", + env=env, + model_path=qwen3, cwd=llm_venv.get_working_directory()) -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) def test_disaggregated_benchmark_gen_only_insufficient_kv( - disaggregated_test_root, disaggregated_example_root, llm_venv, - llama_model_root): + disaggregated_test_root, disaggregated_example_root, llm_venv): """Test that gen-only benchmark mode raises an error when KV cache is too small to hold all benchmark requests, instead of hanging forever.""" import openai - setup_model_symlink(llm_venv, llama_model_root, - "TinyLlama/TinyLlama-1.1B-Chat-v1.0") + qwen3 = _qwen3_model_root() + setup_model_symlink(llm_venv, qwen3, _QWEN3_HF_ID) env = llm_venv._new_env.copy() env['TRTLLM_DISAGG_BENCHMARK_GEN_ONLY'] = '1' @@ -1257,7 +1238,7 @@ def test_disaggregated_benchmark_gen_only_insufficient_kv( os.path.dirname(__file__)) config, ctx_workers, gen_workers, disagg_server, server_port, work_dir = \ setup_disagg_cluster(config_file, - model_name=llama_model_root, + model_name=qwen3, env=env, cwd=llm_venv.get_working_directory()) @@ -1272,7 +1253,7 @@ def test_disaggregated_benchmark_gen_only_insufficient_kv( def send_request(): try: stream = client.completions.create( - model="TinyLlama/TinyLlama-1.1B-Chat-v1.0", + model=_QWEN3_HF_ID, prompt="What is the capital of Germany?", max_tokens=10, temperature=0.0, @@ -1298,147 +1279,103 @@ def send_request(): @pytest.mark.skip_less_device(4) -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) def test_disaggregated_genbs1(disaggregated_test_root, - disaggregated_example_root, llm_venv, - llama_model_root): - setup_model_symlink(llm_venv, llama_model_root, - "TinyLlama/TinyLlama-1.1B-Chat-v1.0") + disaggregated_example_root, llm_venv): + qwen3 = _qwen3_model_root() + setup_model_symlink(llm_venv, qwen3, _QWEN3_HF_ID) env = llm_venv._new_env.copy() env['TRTLLM_DISAGG_BENCHMARK_GEN_ONLY'] = '1' run_disaggregated_test(disaggregated_example_root, "gen_only_bs1", env=env, - model_path=llama_model_root, + model_path=qwen3, cwd=llm_venv.get_working_directory()) -@pytest.mark.skip_less_device(2) -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) -def test_disaggregated_multi_gpu(disaggregated_test_root, - disaggregated_example_root, llm_venv, - llama_model_root): - setup_model_symlink(llm_venv, llama_model_root, - "TinyLlama/TinyLlama-1.1B-Chat-v1.0") +def test_disaggregated_multi_orchestrator(disaggregated_test_root, + disaggregated_example_root, llm_venv): + qwen3 = _qwen3_model_root() + setup_model_symlink(llm_venv, qwen3, _QWEN3_HF_ID) + env = llm_venv._new_env.copy() + env["CUDA_VISIBLE_DEVICES"] = "0" run_disaggregated_test(disaggregated_example_root, - "4_ranks", - env=llm_venv._new_env, - model_path=llama_model_root, + "multi_orchestrator", + num_iters=1, + env=env, + model_path=qwen3, cwd=llm_venv.get_working_directory()) -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) def test_disaggregated_cuda_graph(disaggregated_test_root, llm_venv, - disaggregated_example_root, llama_model_root): - setup_model_symlink(llm_venv, llama_model_root, - "TinyLlama/TinyLlama-1.1B-Chat-v1.0") + disaggregated_example_root): + qwen3 = _qwen3_model_root() + setup_model_symlink(llm_venv, qwen3, _QWEN3_HF_ID) run_disaggregated_test(disaggregated_example_root, "cuda_graph", env=llm_venv._new_env, - model_path=llama_model_root, + model_path=qwen3, cwd=llm_venv.get_working_directory()) -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) def test_disaggregated_mixed(disaggregated_test_root, llm_venv, - disaggregated_example_root, llama_model_root): - setup_model_symlink(llm_venv, llama_model_root, - "TinyLlama/TinyLlama-1.1B-Chat-v1.0") + disaggregated_example_root): + qwen3 = _qwen3_model_root() + setup_model_symlink(llm_venv, qwen3, _QWEN3_HF_ID) run_disaggregated_test(disaggregated_example_root, "mixed", env=llm_venv._new_env, - model_path=llama_model_root, - cwd=llm_venv.get_working_directory()) - - -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) -def test_disaggregated_overlap(disaggregated_test_root, llm_venv, - disaggregated_example_root, llama_model_root): - setup_model_symlink(llm_venv, llama_model_root, - "TinyLlama/TinyLlama-1.1B-Chat-v1.0") - - def post_client_test(server_url: str): - verify_usage_with_cache_reuse(server_url, - "TinyLlama/TinyLlama-1.1B-Chat-v1.0") - - run_disaggregated_test(disaggregated_example_root, - "overlap", - env=llm_venv._new_env, - post_client_test=post_client_test, - model_path=llama_model_root, + model_path=qwen3, cwd=llm_venv.get_working_directory()) @skip_pre_hopper @pytest.mark.skip_less_device(8) -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) @pytest.mark.parametrize("ctx_pp", [1, 4], ids=["ctx_pp1", "ctx_pp4"]) def test_disaggregated_overlap_gen_first(disaggregated_test_root, disaggregated_example_root, llm_venv, - llama_model_root, ctx_pp): - src_dst_dict = { - llama_model_root: - f"{llm_venv.get_working_directory()}/TinyLlama/TinyLlama-1.1B-Chat-v1.0", - } - for src, dst in src_dst_dict.items(): - if not os.path.islink(dst): - os.makedirs(os.path.dirname(dst), exist_ok=True) - os.symlink(src, dst, target_is_directory=True) + ctx_pp): + qwen3 = _qwen3_model_root() + setup_model_symlink(llm_venv, qwen3, _QWEN3_HF_ID) def post_client_test(server_url: str): - verify_usage_with_cache_reuse(server_url, - "TinyLlama/TinyLlama-1.1B-Chat-v1.0") + verify_usage_with_cache_reuse(server_url, _QWEN3_HF_ID) run_disaggregated_test( disaggregated_example_root, "overlap_gen_first" if ctx_pp == 1 else "overlap_gen_first_pp4", env=llm_venv._new_env, - model_path=llama_model_root, + model_path=qwen3, cwd=llm_venv.get_working_directory(), disagg_schedule_style="generation_first", post_client_test=post_client_test) -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) def test_disaggregated_overlap_transceiver_runtime_python( - disaggregated_test_root, llm_venv, disaggregated_example_root, - llama_model_root): - setup_model_symlink(llm_venv, llama_model_root, - "TinyLlama/TinyLlama-1.1B-Chat-v1.0") + disaggregated_test_root, llm_venv, disaggregated_example_root): + qwen3 = _qwen3_model_root() + setup_model_symlink(llm_venv, qwen3, _QWEN3_HF_ID) env = llm_venv._new_env.copy() env["UCX_TLS"] = get_ucx_tls() run_disaggregated_test(disaggregated_example_root, "overlap_transceiver_runtime_python", env=env, - model_path=llama_model_root, + model_path=qwen3, cwd=llm_venv.get_working_directory()) -# Exercises the disaggregated KV-cache transfer path with the Python cache transceiver runtime -# while the KV-cache pool itself is allocated from fabric (MNNVL) VMM memory via -# TRTLLM_KVCACHE_POOL_USE_FABRIC_MEMORY=1. Restricted to GB200/GB300 since those are the only -# platforms with MNNVL fabric-memory support; on other devices the env var would silently fall -# back to a non-fabric allocation, which would defeat the purpose of this test. +# Exercises the KV-cache transfer path with the Python transceiver while the +# KV-cache pool is allocated from fabric (MNNVL) VMM memory via +# TRTLLM_KVCACHE_POOL_USE_FABRIC_MEMORY=1. Restricted to GB200/GB300. @pytest.mark.skip_device_not_contain(["GB200", "GB300"]) -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) def test_disaggregated_overlap_transceiver_runtime_python_fabric_memory( - disaggregated_test_root, llm_venv, disaggregated_example_root, - llama_model_root): - setup_model_symlink(llm_venv, llama_model_root, - "TinyLlama/TinyLlama-1.1B-Chat-v1.0") + disaggregated_test_root, llm_venv, disaggregated_example_root): + qwen3 = _qwen3_model_root() + setup_model_symlink(llm_venv, qwen3, _QWEN3_HF_ID) env = llm_venv._new_env.copy() env["UCX_TLS"] = get_ucx_tls() @@ -1446,40 +1383,26 @@ def test_disaggregated_overlap_transceiver_runtime_python_fabric_memory( run_disaggregated_test(disaggregated_example_root, "overlap_transceiver_runtime_python", env=env, - model_path=llama_model_root, + model_path=qwen3, cwd=llm_venv.get_working_directory()) -# Exercises the disaggregated KV-cache transfer path with the Python cache transceiver AND the -# KV-cache bounce optimization (cache_transceiver_config.kv_cache_bounce_size_mb > 0): scattered -# per-block WRITEs are gathered into one coalesced fabric-VMM buffer before a single NIXL WRITE. -# Restricted to GB200/GB300 since the bounce arena is fabric (MNNVL) VMM memory. -# -# The bounce transport coalesces a transfer only when it clears the receiver's min_blocks gate. -# The test lowers that gate via the TRTLLM_KV_CACHE_BOUNCE_MIN_BLOCKS env so the ordinary short -# prompts still take the coalesced-bounce WRITE path -- no special long prompt is needed. The test -# runs the normal disagg output verification (the coalesced KV must still decode to the right -# answer, e.g. "Berlin"; a corrupt transfer would garble it) AND asserts the generation worker -# logged the coalesced-bounce marker, so a silent fall-back to the per-fragment path fails the -# test instead of passing quietly. +# Exercises the KV-cache bounce-buffer path (kv_cache_bounce_size_mb > 0). +# Restricted to GB200/GB300 since the bounce arena requires MNNVL fabric VMM. @pytest.mark.skip_device_not_contain(["GB200", "GB300"]) -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) def test_disaggregated_overlap_transceiver_runtime_python_bounce( - disaggregated_test_root, llm_venv, disaggregated_example_root, - llama_model_root): - setup_model_symlink(llm_venv, llama_model_root, - "TinyLlama/TinyLlama-1.1B-Chat-v1.0") + disaggregated_test_root, llm_venv, disaggregated_example_root): + qwen3 = _qwen3_model_root() + setup_model_symlink(llm_venv, qwen3, _QWEN3_HF_ID) env = llm_venv._new_env.copy() env["UCX_TLS"] = get_ucx_tls() - # min_blocks=1 forces bounce on even for the short test prompt (the gate is internal, tuned via - # env). No fabric-pool env is needed: the bounce arena is its own fabric memory. + # min_blocks=1 forces bounce even for short prompts. env["TRTLLM_KV_CACHE_BOUNCE_MIN_BLOCKS"] = "1" run_disaggregated_test(disaggregated_example_root, "overlap_transceiver_runtime_python_bounce", env=env, - model_path=llama_model_root, + model_path=qwen3, cwd=llm_venv.get_working_directory(), assert_gen_log_contains="[kv-bounce] coalesced") @@ -1499,27 +1422,14 @@ def _verify_python_transceiver_under_host_offload(server_url: str, model: str): 2. Send more prompts, evicting earlier blocks to the host pool. 3. Re-issue the earlier prompts. Reuse hits force onboard from host back to primary, and the disagg transfer must read primary slots - that no longer match the original block IDs. With the fix, this - succeeds; without it, the sender either crashes on a primary - assertion or returns nonsense tokens. - - Assertions are deliberately content-agnostic (TinyLlama outputs vary - run-to-run): we check that responses are non-empty, the server stays - up across the eviction/onboard cycle, and `cached_tokens > 0` on - repeats so we know reuse actually fired. + that no longer match the original block IDs. + + Assertions are content-agnostic: we check that responses are non-empty, + the server stays up across the eviction/onboard cycle, and + `cached_tokens > 0` on repeats so we know reuse actually fired. """ timeout = aiohttp.ClientTimeout(total=180) max_tokens = 16 - # Workload sizing: ctx-side primary pool = max_tokens(1024) / - # tokens_per_block(64) = 16 blocks. Each prompt below tokenizes to - # ~200 tokens ≈ 4 KV blocks. We send 6 distinct prompts → ~24 blocks - # of primary demand > 16-block primary pool, forcing eviction of an - # earlier prefix to host. Replaying earlier prompts (Pass 2) then - # forces onboard from host back to primary, and onboard typically - # places the block in a *different* primary slot than its block_id. - # That divergence is exactly what the disagg pointer-arithmetic fix - # has to handle — without the fix, the sender computes - # `base + block_id * slot_bytes` and reads the wrong primary slot. _filler = ( "This is filler context describing computer systems, distributed " "inference, KV cache management, host memory offload policies, " @@ -1567,20 +1477,13 @@ def assert_sane(resp, label): async def drive(): async with aiohttp.ClientSession() as session: - # Pass 1: prime the radix tree with each distinct prompt and - # capture the deterministic output (temperature=0). + # Pass 1: prime the radix tree with each distinct prompt. first_texts = [] for idx, p in enumerate(distinct_prompts): resp = await send(session, p) first_texts.append(assert_sane(resp, f"pass1[{idx}]")) - # Pass 2: send all prompts CONCURRENTLY each replay. Concurrent - # in-flight prefills hold their KV blocks simultaneously; with - # primary capacity smaller than the union of in-flight prompts, - # this is the scenario that produces non-trivial alloc/free - # interleaving and onboard-to-different-slot for replayed - # prompts. Strict serial sends (Pass 1 above) typically alloc - # back to original slots and miss the bug. + # Pass 2: concurrent replays to stress onboard-to-different-slot. for replay in range(5): results = await asyncio.gather( *[send(session, p) for p in distinct_prompts]) @@ -1593,13 +1496,9 @@ async def drive(): print(f"[host_offload_e2e] replay={replay} prompt={idx} " f"prompt_tokens={usage.get('prompt_tokens')} " f"cached_tokens={cached}") - # Reuse must hit — otherwise we never exercise onboard - # back from host, which is the path the fix protects. assert cached > 0, ( f"replay={replay} prompt={idx}: expected reuse " f"hit (cached_tokens > 0), got usage={usage}") - # Primary regression check: deterministic decoding + - # correct KV must reproduce Pass 1's output bit-for-bit. assert text == first_texts[idx], ( f"replay={replay} prompt={idx}: output diverged " f"from Pass 1, indicating wrong KV was read after " @@ -1610,14 +1509,8 @@ async def drive(): asyncio.run(drive()) -# Plain parametrize (not the `llama_model_root` indirect fixture) so the -# test ID picks up the `[TinyLlama-1.1B-Chat-v1.0]` suffix that matches -# the other disagg tests, without forcing LLM_MODELS_ROOT / NFS access — -# trtllm-serve resolves the HuggingFace id directly. -@pytest.mark.parametrize("llama_model_root", ["TinyLlama-1.1B-Chat-v1.0"]) def test_disaggregated_python_transceiver_host_offload( - disaggregated_test_root, llm_venv, disaggregated_example_root, - llama_model_root): # noqa: ARG001 — used only for the parametrize label + disaggregated_test_root, llm_venv, disaggregated_example_root): """E2E regression for block_id -> primary-slot translation in the Python disagg cache transceiver. See `_verify_python_transceiver_under_host_offload` for what this @@ -1625,35 +1518,27 @@ def test_disaggregated_python_transceiver_host_offload( ctx-side `host_cache_size` and a deliberately tight primary pool so that prefix reuse is forced through an offload+onboard cycle before each KV transfer. - - Model resolution: trtllm-serve loads the HuggingFace id from the - config's `model:` field (TinyLlama/TinyLlama-1.1B-Chat-v1.0) via - huggingface_hub on first use. No LLM_MODELS_ROOT / NFS dependency. """ - setup_model_symlink(llm_venv, llama_model_root, - "TinyLlama/TinyLlama-1.1B-Chat-v1.0") + qwen3 = _qwen3_model_root() + setup_model_symlink(llm_venv, qwen3, _QWEN3_HF_ID) env = llm_venv._new_env.copy() env["UCX_TLS"] = get_ucx_tls() def post_client_test(server_url: str): - _verify_python_transceiver_under_host_offload( - server_url, "TinyLlama/TinyLlama-1.1B-Chat-v1.0") + _verify_python_transceiver_under_host_offload(server_url, _QWEN3_HF_ID) run_disaggregated_test(disaggregated_example_root, "python_transceiver_host_offload", env=env, - model_path=llama_model_root, + model_path=qwen3, cwd=llm_venv.get_working_directory(), post_client_test=post_client_test) -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) def test_disaggregated_perf_metrics(disaggregated_test_root, llm_venv, - disaggregated_example_root, - llama_model_root, tmp_path): - setup_model_symlink(llm_venv, llama_model_root, - "TinyLlama/TinyLlama-1.1B-Chat-v1.0") + disaggregated_example_root, tmp_path): + qwen3 = _qwen3_model_root() + setup_model_symlink(llm_venv, qwen3, _QWEN3_HF_ID) perf_metrics_output_dir = str(tmp_path / "perf_metrics") @@ -1673,36 +1558,15 @@ def extra_endpoints_test(_server_url: str): "perf_metrics", env=env, extra_endpoints_test=extra_endpoints_test, - model_path=llama_model_root, + model_path=qwen3, cwd=llm_venv.get_working_directory(), perf_metrics_output_dir=perf_metrics_output_dir) -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) -def test_disaggregated_chat_completion_tool_calls(disaggregated_test_root, - llm_venv, - disaggregated_example_root, - llama_model_root): - setup_model_symlink(llm_venv, llama_model_root, - "TinyLlama/TinyLlama-1.1B-Chat-v1.0") - - run_disaggregated_test(disaggregated_example_root, - "tool_calls", - num_iters=1, - prompt_file="tool_call_prompts.json", - env=llm_venv._new_env, - model_path=llama_model_root, - cwd=llm_venv.get_working_directory()) - - -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) def test_disaggregated_kv_cache_time_output(disaggregated_test_root, llm_venv, - disaggregated_example_root, - llama_model_root): - setup_model_symlink(llm_venv, llama_model_root, - "TinyLlama/TinyLlama-1.1B-Chat-v1.0") + disaggregated_example_root): + qwen3 = _qwen3_model_root() + setup_model_symlink(llm_venv, qwen3, _QWEN3_HF_ID) output_path = os.path.join(llm_venv.get_working_directory(), "cache_time") env = llm_venv._new_env.copy() @@ -1714,8 +1578,9 @@ def test_disaggregated_kv_cache_time_output(disaggregated_test_root, llm_venv, env["TRTLLM_KVCACHE_TIME_OUTPUT_PATH"] = output_path run_disaggregated_test(disaggregated_example_root, "perf_metrics", - env=env, - model_path=llama_model_root, + env=env + | {"TRTLLM_KVCACHE_TIME_OUTPUT_PATH": output_path}, + model_path=qwen3, cwd=llm_venv.get_working_directory()) assert os.path.isdir(output_path) # The C++ transceiver names timing files "__.csv" @@ -1753,155 +1618,95 @@ def test_disaggregated_kv_cache_time_output(disaggregated_test_root, llm_venv, assert matched -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) def test_disaggregated_load_balance(disaggregated_test_root, llm_venv, - disaggregated_example_root, - llama_model_root): - setup_model_symlink(llm_venv, llama_model_root, - "TinyLlama/TinyLlama-1.1B-Chat-v1.0") + disaggregated_example_root): + qwen3 = _qwen3_model_root() + setup_model_symlink(llm_venv, qwen3, _QWEN3_HF_ID) run_disaggregated_test(disaggregated_example_root, "load_balance", env=llm_venv._new_env, - model_path=llama_model_root, + model_path=qwen3, cwd=llm_venv.get_working_directory()) -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) -def test_disaggregated_cache_aware_balance(disaggregated_test_root, llm_venv, - disaggregated_example_root, - llama_model_root): - setup_model_symlink(llm_venv, llama_model_root, - "TinyLlama/TinyLlama-1.1B-Chat-v1.0") - - run_disaggregated_test(disaggregated_example_root, - "cache_aware_balance", - env=llm_venv._new_env, - model_path=llama_model_root, - cwd=llm_venv.get_working_directory()) - - -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) -def test_disaggregated_conditional(disaggregated_test_root, llm_venv, - disaggregated_example_root, - llama_model_root): - setup_model_symlink(llm_venv, llama_model_root, - "TinyLlama/TinyLlama-1.1B-Chat-v1.0") - - run_disaggregated_test(disaggregated_example_root, - "conditional", - env=llm_venv._new_env, - model_path=llama_model_root, - cwd=llm_venv.get_working_directory()) - - -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) def test_disaggregated_ngram(disaggregated_test_root, llm_venv, - disaggregated_example_root, llama_model_root): - setup_model_symlink(llm_venv, llama_model_root, - "TinyLlama/TinyLlama-1.1B-Chat-v1.0") + disaggregated_example_root): + qwen3 = _qwen3_model_root() + setup_model_symlink(llm_venv, qwen3, _QWEN3_HF_ID) run_disaggregated_test(disaggregated_example_root, "ngram", env=llm_venv._new_env, - model_path=llama_model_root, + model_path=qwen3, cwd=llm_venv.get_working_directory()) @pytest.mark.skip_less_device(4) -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) def test_disaggregated_ctxpp2_genpp2(disaggregated_test_root, llm_venv, - disaggregated_example_root, - llama_model_root): - setup_model_symlink(llm_venv, llama_model_root, - "TinyLlama/TinyLlama-1.1B-Chat-v1.0") + disaggregated_example_root): + qwen3 = _qwen3_model_root() + setup_model_symlink(llm_venv, qwen3, _QWEN3_HF_ID) run_disaggregated_test(disaggregated_example_root, "ctxpp2_genpp2", env=llm_venv._new_env, - model_path=llama_model_root, + model_path=qwen3, cwd=llm_venv.get_working_directory()) @pytest.mark.skip_less_device(4) -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) def test_disaggregated_ctxtp2_genpp2(disaggregated_test_root, llm_venv, - disaggregated_example_root, - llama_model_root): - setup_model_symlink(llm_venv, llama_model_root, - "TinyLlama/TinyLlama-1.1B-Chat-v1.0") + disaggregated_example_root): + qwen3 = _qwen3_model_root() + setup_model_symlink(llm_venv, qwen3, _QWEN3_HF_ID) run_disaggregated_test(disaggregated_example_root, "ctxtp2_genpp2", env=llm_venv._new_env, - model_path=llama_model_root, - cwd=llm_venv.get_working_directory()) - - -@pytest.mark.skip_less_device(4) -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) -def test_disaggregated_ctxpp2_gentp2(disaggregated_test_root, llm_venv, - disaggregated_example_root, - llama_model_root): - setup_model_symlink(llm_venv, llama_model_root, - "TinyLlama/TinyLlama-1.1B-Chat-v1.0") - run_disaggregated_test(disaggregated_example_root, - "ctxpp2_gentp2", - env=llm_venv._new_env, - model_path=llama_model_root, + model_path=qwen3, cwd=llm_venv.get_working_directory()) @pytest.mark.skip_less_device(8) -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) def test_disaggregated_ctxtp2pp2_gentp2pp2(disaggregated_test_root, llm_venv, - disaggregated_example_root, - llama_model_root): - setup_model_symlink(llm_venv, llama_model_root, - "TinyLlama/TinyLlama-1.1B-Chat-v1.0") + disaggregated_example_root): + qwen3 = _qwen3_model_root() + setup_model_symlink(llm_venv, qwen3, _QWEN3_HF_ID) run_disaggregated_test(disaggregated_example_root, "ctxtp2pp2_gentp2pp2", env=llm_venv._new_env, - model_path=llama_model_root, + model_path=qwen3, cwd=llm_venv.get_working_directory()) @pytest.mark.skip_less_device(8) -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) def test_disaggregated_ctxpp4_genpp4(disaggregated_test_root, llm_venv, - disaggregated_example_root, - llama_model_root): - setup_model_symlink(llm_venv, llama_model_root, - "TinyLlama/TinyLlama-1.1B-Chat-v1.0") + disaggregated_example_root): + qwen3 = _qwen3_model_root() + setup_model_symlink(llm_venv, qwen3, _QWEN3_HF_ID) run_disaggregated_test(disaggregated_example_root, "ctxpp4_genpp4", env=llm_venv._new_env, - model_path=llama_model_root, + model_path=qwen3, cwd=llm_venv.get_working_directory()) -#tiny llama pp4 will have uneven layer per pp. pp4 +# Qwen3-0.6B pp4 will have uneven layers per pp rank; pp4 exercises that code path. @pytest.mark.skip_less_device(8) -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) def test_disaggregated_ctxpp4_gentp4(disaggregated_test_root, llm_venv, - disaggregated_example_root, - llama_model_root): - setup_model_symlink(llm_venv, llama_model_root, - "TinyLlama/TinyLlama-1.1B-Chat-v1.0") + disaggregated_example_root): + qwen3 = _qwen3_model_root() + setup_model_symlink(llm_venv, qwen3, _QWEN3_HF_ID) run_disaggregated_test(disaggregated_example_root, "ctxpp4_gentp4", env=llm_venv._new_env, - model_path=llama_model_root, + model_path=qwen3, cwd=llm_venv.get_working_directory()) +# --------------------------------------------------------------------------- +# End of Qwen3-0.6B disaggregated tests +# --------------------------------------------------------------------------- + + @skip_no_hopper @pytest.mark.skip_less_device(4) @pytest.mark.skip( @@ -2899,44 +2704,6 @@ def test_disaggregated_deepseek_v3_lite_bf16_empty_batch( assert e2el > 0 and ttft > 0 -@pytest.mark.skip_less_device(8) -@pytest.mark.skip_less_device_memory(140000) -@pytest.mark.parametrize( - "model_path", - ['llama4-models/nvidia/Llama-4-Maverick-17B-128E-Instruct-FP8']) -def test_llama4_long_context_kv_cache_overflow(disaggregated_test_root, - disaggregated_example_root, - llm_venv, model_path): - """ - RCCA: https://nvbugspro.nvidia.com/bug/5555681 - Test to reproduce KV cache buffer overflow bug with long context. - """ - models_root = llm_models_root() - llama4_model_root = os.path.join(models_root, model_path) - - # Create symlink to match config file path - setup_model_symlink(llm_venv, llama4_model_root, model_path) - - config_file = get_test_config("llama4_kv_cache_overflow", - disaggregated_example_root, - os.path.dirname(__file__)) - - run_disaggregated_aiperf( - config_file=config_file, - model_path=llama4_model_root, - server_start_timeout=1200, - input_tokens=128000, - output_tokens=100, - # This repro intentionally degrades the KV - # transfer path (tiny max_tokens_in_buffer vs - # 128k inputs), so sporadic request errors are - # by-design; keep the test scoped to its - # original crash/fatal-log checks. - max_error_rate=None, - env=llm_venv._new_env, - cwd=llm_venv.get_working_directory()) - - @skip_pre_blackwell @pytest.mark.timeout(2400) @pytest.mark.skip_less_device(4) @@ -4224,6 +3991,66 @@ def test_disaggregated_cancel_large_context_requests_long( cwd=llm_venv.get_working_directory()) +def _verify_mamba_bs1_concurrency2(server_url: str) -> None: + + async def run() -> None: + timeout = aiohttp.ClientTimeout(total=120) + prompts = ( + "Write one sentence about disaggregated inference.", + "Write one sentence about recurrent-state transfer.", + ) + + async def send(session: aiohttp.ClientSession, prompt: str) -> None: + payload = { + "model": MAMBA_BS1_CONCURRENCY2_MODEL, + "prompt": prompt, + "max_tokens": 16, + "temperature": 0, + "ignore_eos": True, + } + async with session.post(f"{server_url}/v1/completions", + json=payload, + timeout=timeout) as response: + body = await response.json() + assert response.status == 200, body + assert body.get("choices"), body + + async with aiohttp.ClientSession() as session: + await asyncio.gather(*(send(session, prompt) for prompt in prompts)) + + asyncio.run(run()) + + +@skip_pre_blackwell +@pytest.mark.timeout(900) +def test_disaggregated_mamba_bs1_concurrency2(disaggregated_example_root, + llm_venv): + model_path = f"{llm_models_root()}/{MAMBA_BS1_CONCURRENCY2_MODEL}" + env = llm_venv._new_env.copy() + repo_root = os.path.abspath( + os.path.join(os.path.dirname(__file__), "../../../..")) + env["LLM_ROOT"] = repo_root + env["PYTHONPATH"] = os.pathsep.join(path for path in (repo_root, + env.get("PYTHONPATH")) + if path) + env.pop("TRTLLM_DISABLE_KV_CACHE_TRANSFER_OVERLAP", None) + env["TRTLLM_NIXL_NUM_THREADS"] = "1" + worker_env = {"TRTLLM_DISAGG_BENCHMARK_GEN_ONLY": "1"} + run_disaggregated_test( + disaggregated_example_root, + "mamba_bs1_concurrency2", + num_iters=0, + env=env, + model_path=model_path, + cwd=llm_venv.get_working_directory(), + post_client_test=_verify_mamba_bs1_concurrency2, + ctx_env=worker_env, + gen_env=worker_env, + share_gpu=True, + server_start_timeout=600, + ) + + @pytest.mark.skip_less_device(8) @skip_pre_blackwell @pytest.mark.parametrize("model_path", diff --git a/tests/integration/defs/disaggregated/test_disaggregated_etcd.py b/tests/integration/defs/disaggregated/test_disaggregated_etcd.py index a3a047a93652..04c1aa9fb9e4 100644 --- a/tests/integration/defs/disaggregated/test_disaggregated_etcd.py +++ b/tests/integration/defs/disaggregated/test_disaggregated_etcd.py @@ -311,7 +311,7 @@ def run_automated_disaggregated_test(example_dir, env=None, cwd=None): kill_automated_disaggregated_processes() cleanup_automated_output_files() - config = {"model_path": "TinyLlama/TinyLlama-1.1B-Chat-v1.0"} + config = {"model_path": "Qwen3/Qwen3-0.6B"} # Create configuration files create_config_files(config) @@ -427,14 +427,13 @@ def run_automated_disaggregated_test(example_dir, env=None, cwd=None): kill_automated_disaggregated_processes() -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) +@pytest.mark.parametrize("llama_model_root", ['Qwen3-0.6B'], indirect=True) def test_automated_disaggregated_complete(disaggregated_test_root, disaggregated_example_root, llm_venv, llama_model_root): src_dst_dict = { llama_model_root: - f"{llm_venv.get_working_directory()}/TinyLlama/TinyLlama-1.1B-Chat-v1.0", + f"{llm_venv.get_working_directory()}/Qwen3/Qwen3-0.6B", } for src, dst in src_dst_dict.items(): if not os.path.islink(dst): diff --git a/tests/integration/defs/disaggregated/test_disaggregated_single_gpu.py b/tests/integration/defs/disaggregated/test_disaggregated_single_gpu.py index beec99cff550..9d5528f47000 100644 --- a/tests/integration/defs/disaggregated/test_disaggregated_single_gpu.py +++ b/tests/integration/defs/disaggregated/test_disaggregated_single_gpu.py @@ -48,10 +48,10 @@ def get_ucx_tls(): MODEL_PATHS = { "DeepSeek-V3-Lite-fp8": "DeepSeek-V3-Lite/fp8", - "TinyLlama-1.1B-Chat-v1.0": "llama-models-v2/TinyLlama-1.1B-Chat-v1.0", "Llama-3.1-8B-Instruct": "llama-3.1-model/Llama-3.1-8B-Instruct/", "EAGLE3-LLaMA3.1-Instruct-8B": "EAGLE3-LLaMA3.1-Instruct-8B", "Qwen3-8B-FP8": "Qwen3/Qwen3-8B-FP8", + "Qwen3-0.6B": "Qwen3/Qwen3-0.6B", } @@ -359,21 +359,6 @@ def verify_disaggregated(model, generation_overlap, enable_cuda_graph, prompt, print("All workers terminated.") -@pytest.mark.parametrize("model", ["TinyLlama-1.1B-Chat-v1.0"]) -@pytest.mark.parametrize("generation_overlap", [False, True]) -@pytest.mark.parametrize("enable_cuda_graph", [False, True]) -def test_disaggregated_simple_llama(model, generation_overlap, - enable_cuda_graph): - verify_disaggregated( - model, generation_overlap, enable_cuda_graph, - "What is the capital of Germany?", - "\n<|assistant|>\nThe capital of Germany is Berlin. \n<|user|>", [ - 2, 29871, 13, 29966, 29989, 465, 22137, 29989, 29958, 13, 1576, - 7483, 310, 9556, 338, 5115, 29889, 2, 29871, 13, 29966, 29989, 1792, - 29989, 29958 - ]) - - @skip_no_hopper @pytest.mark.parametrize("model", ["DeepSeek-V3-Lite-fp8/fp8"]) @pytest.mark.parametrize("generation_overlap", [False, True]) @@ -622,356 +607,7 @@ def test_disaggregated_spec_dec_batch_slot_limit(model, spec_dec_model_path, print("All workers terminated.") -@pytest.mark.parametrize("model", ["TinyLlama-1.1B-Chat-v1.0"]) -@pytest.mark.parametrize("generation_overlap", [False, True]) -def test_disaggregated_logprobs(model, generation_overlap): - """Verify that logprobs propagate correctly from prefill to decode. - - Ensures first_gen_log_probs is carried in DisaggregatedParams - so the generation_only worker receives one logprob per token. - """ - worker_pytorch_configs = [ - dict(disable_overlap_scheduler=True), - dict(disable_overlap_scheduler=not generation_overlap), - ] - - kv_cache_configs = [KvCacheConfig(max_tokens=2048 * 8) for _ in range(2)] - cache_transceiver_configs = [ - CacheTransceiverConfig(backend="DEFAULT") for _ in range(2) - ] - model_names = [model_path(model) for _ in range(2)] - ranks = [0, 1] - worker_args = list( - zip(kv_cache_configs, cache_transceiver_configs, worker_pytorch_configs, - model_names, ranks)) - - port_name = mpi_publish_name() - max_tokens = 10 - prompt = "What is the capital of Germany?" - - with MPIPoolExecutor(max_workers=2, - env={ - "UCX_TLS": get_ucx_tls(), - "UCX_MM_ERROR_HANDLING": "y" - }) as executor: - futures = [] - try: - for worker_arg in worker_args: - future = executor.submit(worker_entry_point, *worker_arg) - futures.append(future) - except Exception as e: - print(f"Error in worker {worker_arg}: {e}") - raise e - - intercomm = None - try: - intercomm = mpi_initialize_intercomm(port_name) - for _ in range(2): - intercomm.recv(tag=MPI_READY) - - # --- Context-only phase (prefill) with logprobs --- - ctx_requests = [(prompt, - SamplingParams(max_tokens=max_tokens, - ignore_eos=True, - logprobs=1), - DisaggregatedParams(request_type="context_only"))] - - ctx_responses = send_requests_to_worker(ctx_requests, 0, intercomm) - ctx_output = ctx_responses[0][0] - - assert ctx_output.disaggregated_params is not None - assert ctx_output.disaggregated_params.request_type == "context_only" - assert len(ctx_output.token_ids) == 1 - - # The context phase must populate first_gen_log_probs. - dp = ctx_output.disaggregated_params - assert dp.first_gen_log_probs is not None, ( - "first_gen_log_probs should be populated by the context phase") - assert len(dp.first_gen_log_probs) >= 1 - for lp_entry in dp.first_gen_log_probs: - assert isinstance(lp_entry, dict) - for token_id, logprob_obj in lp_entry.items(): - assert isinstance(token_id, int) - assert logprob_obj.logprob <= 0.0, ( - "Log probabilities must be non-positive") - - # --- Generation-only phase (decode) with logprobs --- - dp.request_type = "generation_only" - gen_requests = [(prompt, - SamplingParams(max_tokens=max_tokens, - ignore_eos=True, - logprobs=1), dp)] - - gen_responses = send_requests_to_worker(gen_requests, 1, intercomm) - gen_output = gen_responses[0][0] - - # Without first_gen_log_probs propagation this either crashes - # (AttributeError) or returns fewer logprobs than tokens. - assert gen_output.logprobs is not None, ( - "Generation phase should return logprobs") - assert len(gen_output.logprobs) == len(gen_output.token_ids), ( - f"Expected one logprob per token: got {len(gen_output.logprobs)}" - f" logprobs for {len(gen_output.token_ids)} tokens") - - for pos_idx, lp_entry in enumerate(gen_output.logprobs): - assert isinstance( - lp_entry, dict), (f"logprobs[{pos_idx}] should be a dict") - for token_id, logprob_obj in lp_entry.items(): - assert isinstance(token_id, int) - assert logprob_obj.logprob <= 0.0 - - except Exception as e: - print(f"Exception encountered: {e}", flush=True) - raise e - finally: - mpi_send_termination_request(intercomm) - for future in futures: - future.result() - - -@pytest.mark.parametrize("model", ["TinyLlama-1.1B-Chat-v1.0"]) -def test_disaggregated_cancel_gen_requests(model): - # Test that cancelling generation requests on a saturated generation - # worker completes without hangs or resource leaks. - worker_pytorch_configs = [] - - # Context worker - worker_pytorch_configs.append( - dict(disable_overlap_scheduler=True, cuda_graph_config=None)) - - # Generation worker - worker_pytorch_configs.append(dict(cuda_graph_config=None)) - - kv_cache_configs = [ - KvCacheConfig(max_tokens=2048, enable_block_reuse=False) - for _ in range(2) - ] - cache_transceiver_configs = [ - CacheTransceiverConfig(backend="DEFAULT") for _ in range(2) - ] - model_names = [model_path(model) for _ in range(2)] - - port_name = mpi_publish_name() - - prompt = "What is the capital of Germany?" - num_requests = 16 - num_cancel = 8 - max_tokens = 50 - - with MPIPoolExecutor(max_workers=2, - env={ - "UCX_TLS": get_ucx_tls(), - "UCX_MM_ERROR_HANDLING": "y", - }) as executor: - futures = [] - try: - futures.append( - executor.submit(worker_entry_point, kv_cache_configs[0], - cache_transceiver_configs[0], - worker_pytorch_configs[0], model_names[0], 0)) - futures.append( - executor.submit(worker_entry_point, kv_cache_configs[1], - cache_transceiver_configs[1], - worker_pytorch_configs[1], model_names[1], 1, - True)) - except Exception as e: - print(f"Error submitting workers: {e}") - raise e - - intercomm = None - try: - print("Launched all workers.", flush=True) - intercomm = mpi_initialize_intercomm(port_name) - - for _ in range(2): - intercomm.recv(tag=MPI_READY) - print("Received ready signal.") - - context_requests = [] - for _ in range(num_requests): - context_requests.append( - (prompt, SamplingParams(max_tokens=1, ignore_eos=True), - DisaggregatedParams(request_type="context_only"))) - - intercomm.send(context_requests, dest=0, tag=MPI_REQUEST) - - gen_requests = [] - for _ in range(num_requests): - output = intercomm.recv(source=0, tag=MPI_RESULT) - assert output[0].disaggregated_params is not None - assert output[ - 0].disaggregated_params.request_type == "context_only" - assert len(output[0].token_ids) == 1 - - disagg_params = output[0].disaggregated_params - disagg_params.request_type = "generation_only" - gen_requests.append( - (prompt, - SamplingParams(max_tokens=max_tokens, - ignore_eos=True), disagg_params)) - - intercomm.send(gen_requests, dest=1, tag=MPI_REQUEST) - - num_started = intercomm.recv(source=1, tag=MPI_STARTED) - assert num_started == num_requests - print(f"Generation worker started {num_started} requests.") - - cancel_indices = list(range(num_cancel)) - intercomm.send(cancel_indices, dest=1, tag=MPI_CANCEL) - print(f"Sent cancel for indices {cancel_indices}.") - - for i in range(num_requests): - output = intercomm.recv(source=1, tag=MPI_RESULT) - print(f"Received result {i}/{num_requests}.") - - except Exception as e: - print(f"Exception encountered: {e}", flush=True) - finally: - print("Sending termination request", flush=True) - mpi_send_termination_request(intercomm) - - print("Waiting for all workers to terminate.", flush=True) - for future in futures: - future.result() - print("All workers terminated.") - - -@pytest.mark.parametrize("model", ["TinyLlama-1.1B-Chat-v1.0"]) -@pytest.mark.parametrize("generation_overlap", [False, True]) -def test_disaggregated_logits(model, generation_overlap): - """Verify that generation logits propagate from prefill to decode in disagg.""" - worker_pytorch_configs = [] - - # Context worker - worker_pytorch_configs.append(dict(disable_overlap_scheduler=True)) - - # Generation worker - worker_pytorch_configs.append( - dict(disable_overlap_scheduler=not generation_overlap)) - - kv_cache_configs = [KvCacheConfig(max_tokens=2048 * 8) for _ in range(2)] - cache_transceiver_configs = [ - CacheTransceiverConfig(backend="DEFAULT") for _ in range(2) - ] - model_names = [model_path(model) for _ in range(2)] - ranks = [0, 1] - worker_args = list( - zip(kv_cache_configs, cache_transceiver_configs, worker_pytorch_configs, - model_names, ranks)) - - port_name = mpi_publish_name() - - prompt = "What is the capital of Germany?" - max_tokens = 10 - - with MPIPoolExecutor(max_workers=2, - env={ - "UCX_TLS": get_ucx_tls(), - "UCX_MM_ERROR_HANDLING": "y" - }) as executor: - futures = [] - try: - for worker_arg in worker_args: - future = executor.submit(worker_entry_point, *worker_arg) - futures.append(future) - except Exception as e: - print(f"Error in worker {worker_arg}: {e}") - raise e - - intercomm = None - try: - print("Launched all the workers.", flush=True) - intercomm = mpi_initialize_intercomm(port_name) - - for _ in range(2): - intercomm.recv(tag=MPI_READY) - print("Received ready signal.") - - # --- Run aggregated request for reference --- - agg_sp = SamplingParams(max_tokens=max_tokens, - ignore_eos=True, - return_generation_logits=True) - agg_requests = [(prompt, agg_sp, None)] - # Use context worker (rank 0) for aggregated request - agg_responses = send_requests_to_worker(agg_requests, 0, intercomm) - agg_output = agg_responses[0][0] - agg_logits = agg_output.generation_logits - assert agg_logits is not None, \ - "Aggregated request should produce generation_logits" - print(f"Aggregated logits shape: {agg_logits.shape}") - - # --- Run disaggregated: context_only --- - ctx_sp = SamplingParams(max_tokens=max_tokens, - ignore_eos=True, - return_generation_logits=True) - ctx_requests = [(prompt, ctx_sp, - DisaggregatedParams(request_type="context_only"))] - ctx_responses = send_requests_to_worker(ctx_requests, 0, intercomm) - ctx_output = ctx_responses[0][0] - dp = ctx_output.disaggregated_params - - assert dp is not None - assert dp.request_type == "context_only" - assert dp.first_gen_logits is not None, \ - "context_only should produce first_gen_logits" - assert len(dp.first_gen_logits) > 0 - print(f"first_gen_logits[0] shape: {dp.first_gen_logits[0].shape}") - - # --- Run disaggregated: generation_only (non-streaming) --- - dp.request_type = "generation_only" - gen_sp = SamplingParams(max_tokens=max_tokens, - ignore_eos=True, - return_generation_logits=True) - gen_requests = [(prompt, gen_sp, dp)] - gen_responses = send_requests_to_worker(gen_requests, 1, intercomm) - gen_output = gen_responses[0][0] - gen_logits = gen_output.generation_logits - - assert gen_logits is not None, \ - "generation_only with first_gen_logits should produce " \ - "generation_logits" - print(f"Disagg gen logits shape: {gen_logits.shape}, " - f"output tokens: {len(gen_output.token_ids)}") - - # Logits should cover all generated tokens (including the - # first token whose logits were transferred from prefill). - assert gen_logits.shape[0] == len(gen_output.token_ids), \ - (f"generation_logits length {gen_logits.shape[0]} != " - f"output token count {len(gen_output.token_ids)}") - - # --- Run disaggregated: generation_only (streaming) --- - # Re-run context_only to get fresh disagg params. - ctx_responses2 = send_requests_to_worker(ctx_requests, 0, intercomm) - dp2 = ctx_responses2[0][0].disaggregated_params - dp2.request_type = "generation_only" - stream_sp = SamplingParams(max_tokens=max_tokens, - ignore_eos=True, - return_generation_logits=True) - stream_requests = [(prompt, stream_sp, dp2, True)] - stream_responses = send_requests_to_worker(stream_requests, 1, - intercomm) - per_chunk_logits = stream_responses[0] - print(f"Streaming per-chunk logits shapes: {per_chunk_logits}") - assert len(per_chunk_logits) > 0, \ - "Expected at least one streaming chunk" - assert per_chunk_logits[0] is not None, \ - ("First streaming chunk should have generation_logits " - "(first_gen_logits from prefill), but got None") - - except Exception as e: - print(f"Exception encountered: {e}", flush=True) - raise e - finally: - print("Sending termination request", flush=True) - mpi_send_termination_request(intercomm) - - print("Waiting for all workers to terminate. ", flush=True) - for future in futures: - future.result() - print("All workers terminated.") - - -@pytest.mark.parametrize("model", ["TinyLlama-1.1B-Chat-v1.0"]) +@pytest.mark.parametrize("model", ["Qwen3-0.6B"]) @pytest.mark.parametrize("generation_overlap", [False]) def test_arbitrary_kv_cache_transfer(model, generation_overlap): """Test KV cache transfer from the reuse tree. @@ -1122,7 +758,7 @@ def test_arbitrary_kv_cache_transfer(model, generation_overlap): print("All workers terminated.") -@pytest.mark.parametrize("model", ["TinyLlama-1.1B-Chat-v1.0"]) +@pytest.mark.parametrize("model", ["Qwen3-0.6B"]) @pytest.mark.parametrize("generation_overlap", [False]) def test_arbitrary_kv_cache_transfer_missing_blocks(model, generation_overlap): """Test that missing-block transfers fail. @@ -1254,3 +890,221 @@ def test_arbitrary_kv_cache_transfer_missing_blocks(model, generation_overlap): if __name__ == "__main__": pytest.main() + + +@pytest.mark.parametrize("model", ["Qwen3-0.6B"]) +@pytest.mark.parametrize("generation_overlap", [False, True]) +def test_disaggregated_logprobs(model, generation_overlap): + """Verify that logprobs propagate correctly from prefill to decode. + + Ensures first_gen_log_probs is carried in DisaggregatedParams + so the generation_only worker receives one logprob per token. + """ + worker_pytorch_configs = [ + dict(disable_overlap_scheduler=True), + dict(disable_overlap_scheduler=not generation_overlap), + ] + + kv_cache_configs = [KvCacheConfig(max_tokens=2048 * 8) for _ in range(2)] + cache_transceiver_configs = [ + CacheTransceiverConfig(backend="DEFAULT") for _ in range(2) + ] + model_names = [model_path(model) for _ in range(2)] + ranks = [0, 1] + worker_args = list( + zip(kv_cache_configs, cache_transceiver_configs, worker_pytorch_configs, + model_names, ranks)) + + port_name = mpi_publish_name() + max_tokens = 10 + prompt = "What is the capital of Germany?" + + with MPIPoolExecutor(max_workers=2, + env={ + "UCX_TLS": get_ucx_tls(), + "UCX_MM_ERROR_HANDLING": "y" + }) as executor: + futures = [] + try: + for worker_arg in worker_args: + future = executor.submit(worker_entry_point, *worker_arg) + futures.append(future) + except Exception as e: + print(f"Error in worker {worker_arg}: {e}") + raise e + + intercomm = None + try: + intercomm = mpi_initialize_intercomm(port_name) + for _ in range(2): + intercomm.recv(tag=MPI_READY) + + # --- Context-only phase (prefill) with logprobs --- + ctx_requests = [(prompt, + SamplingParams(max_tokens=max_tokens, + ignore_eos=True, + logprobs=1), + DisaggregatedParams(request_type="context_only"))] + + ctx_responses = send_requests_to_worker(ctx_requests, 0, intercomm) + ctx_output = ctx_responses[0][0] + + assert ctx_output.disaggregated_params is not None + assert ctx_output.disaggregated_params.request_type == "context_only" + assert len(ctx_output.token_ids) == 1 + + # The context phase must populate first_gen_log_probs. + dp = ctx_output.disaggregated_params + assert dp.first_gen_log_probs is not None, ( + "first_gen_log_probs should be populated by the context phase") + assert len(dp.first_gen_log_probs) >= 1 + for lp_entry in dp.first_gen_log_probs: + assert isinstance(lp_entry, dict) + for token_id, logprob_obj in lp_entry.items(): + assert isinstance(token_id, int) + assert logprob_obj.logprob <= 0.0, ( + "Log probabilities must be non-positive") + + # --- Generation-only phase (decode) with logprobs --- + dp.request_type = "generation_only" + gen_requests = [(prompt, + SamplingParams(max_tokens=max_tokens, + ignore_eos=True, + logprobs=1), dp)] + + gen_responses = send_requests_to_worker(gen_requests, 1, intercomm) + gen_output = gen_responses[0][0] + + # Without first_gen_log_probs propagation this either crashes + # (AttributeError) or returns fewer logprobs than tokens. + assert gen_output.logprobs is not None, ( + "Generation phase should return logprobs") + assert len(gen_output.logprobs) == len(gen_output.token_ids), ( + f"Expected one logprob per token: got {len(gen_output.logprobs)}" + f" logprobs for {len(gen_output.token_ids)} tokens") + + for pos_idx, lp_entry in enumerate(gen_output.logprobs): + assert isinstance( + lp_entry, dict), (f"logprobs[{pos_idx}] should be a dict") + for token_id, logprob_obj in lp_entry.items(): + assert isinstance(token_id, int) + assert logprob_obj.logprob <= 0.0 + + except Exception as e: + print(f"Exception encountered: {e}", flush=True) + raise e + finally: + mpi_send_termination_request(intercomm) + for future in futures: + future.result() + + +@pytest.mark.parametrize("model", ["Qwen3-0.6B"]) +def test_disaggregated_cancel_gen_requests(model): + # Test that cancelling generation requests on a saturated generation + # worker completes without hangs or resource leaks. + worker_pytorch_configs = [] + + # Context worker + worker_pytorch_configs.append( + dict(disable_overlap_scheduler=True, cuda_graph_config=None)) + + # Generation worker + worker_pytorch_configs.append(dict(cuda_graph_config=None)) + + kv_cache_configs = [ + KvCacheConfig(max_tokens=2048, enable_block_reuse=False) + for _ in range(2) + ] + cache_transceiver_configs = [ + CacheTransceiverConfig(backend="DEFAULT") for _ in range(2) + ] + model_names = [model_path(model) for _ in range(2)] + + port_name = mpi_publish_name() + + prompt = "What is the capital of Germany?" + num_requests = 16 + num_cancel = 8 + max_tokens = 50 + + with MPIPoolExecutor(max_workers=2, + env={ + "UCX_TLS": get_ucx_tls(), + "UCX_MM_ERROR_HANDLING": "y", + }) as executor: + futures = [] + try: + futures.append( + executor.submit(worker_entry_point, kv_cache_configs[0], + cache_transceiver_configs[0], + worker_pytorch_configs[0], model_names[0], 0)) + futures.append( + executor.submit(worker_entry_point, kv_cache_configs[1], + cache_transceiver_configs[1], + worker_pytorch_configs[1], model_names[1], 1, + True)) + except Exception as e: + print(f"Error submitting workers: {e}") + raise e + + intercomm = None + try: + print("Launched all workers.", flush=True) + intercomm = mpi_initialize_intercomm(port_name) + + for _ in range(2): + intercomm.recv(tag=MPI_READY) + print("Received ready signal.") + + context_requests = [] + for _ in range(num_requests): + context_requests.append( + (prompt, SamplingParams(max_tokens=1, ignore_eos=True), + DisaggregatedParams(request_type="context_only"))) + + intercomm.send(context_requests, dest=0, tag=MPI_REQUEST) + + gen_requests = [] + for _ in range(num_requests): + output = intercomm.recv(source=0, tag=MPI_RESULT) + assert output[0].disaggregated_params is not None + assert output[ + 0].disaggregated_params.request_type == "context_only" + assert len(output[0].token_ids) == 1 + + disagg_params = output[0].disaggregated_params + disagg_params.request_type = "generation_only" + gen_requests.append( + (prompt, + SamplingParams(max_tokens=max_tokens, + ignore_eos=True), disagg_params)) + + intercomm.send(gen_requests, dest=1, tag=MPI_REQUEST) + + num_started = intercomm.recv(source=1, tag=MPI_STARTED) + assert num_started == num_requests + print(f"Generation worker started {num_started} requests.") + + cancel_indices = list(range(num_cancel)) + intercomm.send(cancel_indices, dest=1, tag=MPI_CANCEL) + print(f"Sent cancel for indices {cancel_indices}.") + + for i in range(num_requests): + output = intercomm.recv(source=1, tag=MPI_RESULT) + print(f"Received result {i}/{num_requests}.") + + except Exception as e: + print(f"Exception encountered: {e}", flush=True) + finally: + print("Sending termination request", flush=True) + mpi_send_termination_request(intercomm) + + print("Waiting for all workers to terminate.", flush=True) + for future in futures: + future.result() + print("All workers terminated.") + + +if __name__ == "__main__": + pytest.main() diff --git a/tests/integration/defs/disaggregated/test_workers.py b/tests/integration/defs/disaggregated/test_workers.py index 9660178971a8..387d9addb07b 100644 --- a/tests/integration/defs/disaggregated/test_workers.py +++ b/tests/integration/defs/disaggregated/test_workers.py @@ -231,7 +231,7 @@ def __init__(self, gen_servers: List[str], req_timeout_secs: int = DEFAULT_TIMEOUT_REQUEST, server_start_timeout_secs: int = DEFAULT_TIMEOUT_SERVER_START, - model_name: str = "TinyLlama/TinyLlama-1.1B-Chat-v1.0", + model_name: str = "Qwen3/Qwen3-0.6B", internal_request_auth_key: str | None = None): super().__init__(ctx_servers, gen_servers, req_timeout_secs, server_start_timeout_secs, internal_request_auth_key) @@ -285,7 +285,7 @@ def __init__(self, gen_servers: List[str], req_timeout_secs: int = DEFAULT_TIMEOUT_REQUEST, server_start_timeout_secs: int = DEFAULT_TIMEOUT_SERVER_START, - model_name: str = "TinyLlama/TinyLlama-1.1B-Chat-v1.0", + model_name: str = "Qwen3/Qwen3-0.6B", internal_request_auth_key: str | None = None): super().__init__(ctx_servers, gen_servers, req_timeout_secs, server_start_timeout_secs, internal_request_auth_key) @@ -406,7 +406,7 @@ def __init__(self, gen_servers: List[str], req_timeout_secs: int = DEFAULT_TIMEOUT_REQUEST, server_start_timeout_secs: int = DEFAULT_TIMEOUT_SERVER_START, - model_name: str = "TinyLlama/TinyLlama-1.1B-Chat-v1.0", + model_name: str = "Qwen3/Qwen3-0.6B", tokens_per_block: int = 32, internal_request_auth_key: str | None = None): super().__init__(ctx_servers, gen_servers, req_timeout_secs, @@ -555,7 +555,7 @@ async def test_eviction(self): def prepare_llama_model(llama_model_root: str, llm_venv): src_dst_dict = { llama_model_root: - f"{llm_venv.get_working_directory()}/TinyLlama/TinyLlama-1.1B-Chat-v1.0", + f"{llm_venv.get_working_directory()}/Qwen3/Qwen3-0.6B", } for src, dst in src_dst_dict.items(): if not os.path.islink(dst): @@ -677,8 +677,7 @@ def background_workers(llm_venv, config_file: str): @pytest.mark.skip(reason="https://nvbugs/5372970") -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) +@pytest.mark.parametrize("llama_model_root", ['Qwen3-0.6B'], indirect=True) def test_workers_conditional_disaggregation(disaggregated_test_root, disaggregated_example_root, llm_venv, llama_model_root): @@ -725,8 +724,7 @@ def test_workers_conditional_disaggregation_deepseek_v3_lite_bf16( asyncio.run(tester.test_multi_round_request(prompts)) -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) +@pytest.mark.parametrize("llama_model_root", ['Qwen3-0.6B'], indirect=True) def test_workers_kv_cache_events(disaggregated_test_root, disaggregated_example_root, llm_venv, llama_model_root): @@ -745,8 +743,7 @@ def test_workers_kv_cache_events(disaggregated_test_root, asyncio.run(tester.test_multi_round_request(prompts, 6)) -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) +@pytest.mark.parametrize("llama_model_root", ['Qwen3-0.6B'], indirect=True) def test_workers_kv_cache_aware_router(disaggregated_test_root, disaggregated_example_root, llm_venv, llama_model_root): @@ -797,8 +794,7 @@ def test_workers_kv_cache_aware_router_deepseek_v3_lite_bf16( asyncio.run(tester.test_multi_round_request(prompts, 8, 4)) -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) +@pytest.mark.parametrize("llama_model_root", ['Qwen3-0.6B'], indirect=True) def test_workers_kv_cache_aware_router_eviction(disaggregated_test_root, disaggregated_example_root, llm_venv, llama_model_root): @@ -825,7 +821,7 @@ def __init__(self, gen_servers: List[str], req_timeout_secs: int = DEFAULT_TIMEOUT_REQUEST, server_start_timeout_secs: int = DEFAULT_TIMEOUT_SERVER_START, - model_name: str = "TinyLlama/TinyLlama-1.1B-Chat-v1.0", + model_name: str = "Qwen3/Qwen3-0.6B", internal_request_auth_key: str | None = None): super().__init__(ctx_servers, gen_servers, req_timeout_secs, server_start_timeout_secs, internal_request_auth_key) @@ -1003,8 +999,7 @@ async def test_implicit_conversation_matching(self): @skip_no_hopper @pytest.mark.skip_less_device(3) -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) +@pytest.mark.parametrize("llama_model_root", ['Qwen3-0.6B'], indirect=True) def test_workers_conversation_router(disaggregated_test_root, disaggregated_example_root, llm_venv, llama_model_root): diff --git a/tests/integration/defs/examples/test_llm_api_with_mpi.py b/tests/integration/defs/examples/test_llm_api_with_mpi.py index 6be110a9cce0..8fcdb8f6c2fe 100644 --- a/tests/integration/defs/examples/test_llm_api_with_mpi.py +++ b/tests/integration/defs/examples/test_llm_api_with_mpi.py @@ -19,20 +19,25 @@ from defs.common import venv_mpi_check_call -@pytest.mark.parametrize("llama_model_root", ['TinyLlama-1.1B-Chat-v1.0'], - indirect=True) +@pytest.mark.parametrize("llama_model_root", ['Qwen3-0.6B'], indirect=True) def test_llm_api_single_gpu_with_mpirun(llmapi_example_root, llm_venv, llama_model_root): src_dst_dict = { llama_model_root: - f"{llm_venv.get_working_directory()}/TinyLlama/TinyLlama-1.1B-Chat-v1.0", + f"{llm_venv.get_working_directory()}/Qwen3/Qwen3-0.6B", } for src, dst in src_dst_dict.items(): if not os.path.islink(dst): os.makedirs(os.path.dirname(dst), exist_ok=True) os.symlink(src, dst, target_is_directory=True) - summary_cmd = [f"{llmapi_example_root}/quickstart_example.py"] + summary_cmd = [ + "-c", + "from tensorrt_llm import LLM, SamplingParams; " + "llm = LLM(model='Qwen3/Qwen3-0.6B'); " + "out = llm.generate(['A B C'], SamplingParams(max_tokens=4)); " + "assert out[0].outputs[0].token_ids", + ] venv_mpi_check_call(llm_venv, ["mpirun", "-n", "1", "--allow-run-as-root"], summary_cmd) diff --git a/tests/integration/defs/examples/test_ray.py b/tests/integration/defs/examples/test_ray.py index c0f5f84f8fa8..1438f0e40326 100644 --- a/tests/integration/defs/examples/test_ray.py +++ b/tests/integration/defs/examples/test_ray.py @@ -18,6 +18,8 @@ from defs.conftest import get_device_count, llm_models_root from defs.trt_test_alternative import popen +_QWEN3_MODEL = "Qwen3/Qwen3-0.6B" + @pytest.fixture(scope="module") def ray_example_root(llm_root): @@ -27,7 +29,7 @@ def ray_example_root(llm_root): def test_llm_inference_async_ray(ray_example_root, llm_venv): script_path = os.path.join(ray_example_root, "llm_inference_async_ray.py") - model_path = f"{llm_models_root()}/llama-models-v2/TinyLlama-1.1B-Chat-v1.0" + model_path = f"{llm_models_root()}/{_QWEN3_MODEL}" venv_check_call(llm_venv, [script_path, "--model", model_path]) @@ -58,10 +60,9 @@ def test_llm_inference_distributed_ray(ray_example_root, llm_venv, tp_size, if ep_size != -1: model_dir = f"{llm_models_root()}/DeepSeek-V3-Lite/bf16" - cmd.extend(["--model_dir", model_dir]) else: - model_dir = f"{llm_models_root()}/llama-models-v2/TinyLlama-1.1B-Chat-v1.0" - cmd.extend(["--model_dir", model_dir]) + model_dir = f"{llm_models_root()}/{_QWEN3_MODEL}" + cmd.extend(["--model_dir", model_dir]) venv_check_call(llm_venv, cmd) @@ -86,7 +87,7 @@ def _run_ray_disaggregated_serving(ray_example_root, tp_size, disagg_dir = os.path.join(ray_example_root, "disaggregated") script_path = os.path.join(disagg_dir, "disagg_serving_local.sh") - model_dir = f"{llm_models_root()}/llama-models-v2/TinyLlama-1.1B-Chat-v1.0" + model_dir = f"{llm_models_root()}/{_QWEN3_MODEL}" try: runtime_env = { diff --git a/tests/integration/defs/kv_cache/test_final_single_token_context_cuda_graph.py b/tests/integration/defs/kv_cache/test_final_single_token_context_cuda_graph.py index 73522a81991a..d70e6e4ceec3 100644 --- a/tests/integration/defs/kv_cache/test_final_single_token_context_cuda_graph.py +++ b/tests/integration/defs/kv_cache/test_final_single_token_context_cuda_graph.py @@ -31,7 +31,7 @@ from tensorrt_llm._torch.pyexecutor.scheduler import ScheduledRequests from tensorrt_llm._torch.speculative.interface import SpecMetadata from tensorrt_llm._utils import mpi_rank -from tensorrt_llm.executor.executor import GenerationExecutor +from tensorrt_llm.executor import GenerationExecutor from tensorrt_llm.executor.postproc_worker import PostprocWorkerConfig from tensorrt_llm.executor.proxy import GenerationExecutorProxy from tensorrt_llm.executor.worker import GenerationExecutorWorker @@ -41,7 +41,7 @@ from ..conftest import llm_models_root -MODEL = f"{llm_models_root()}/llama-models-v2/TinyLlama-1.1B-Chat-v1.0" +MODEL = f"{llm_models_root()}/Qwen3/Qwen3-0.6B" SPEC_MODEL = f"{llm_models_root()}/Qwen3/Qwen3-8B" EAGLE3_MODEL = f"{llm_models_root()}/Qwen3/qwen3_8b_eagle3" PROMPT_TOKEN_IDS = [1] + [42] * 63 + [43] @@ -360,25 +360,6 @@ def test_final_token_reuse_cuda_graph( _assert_reused_context_used_cuda_graph(graph_executions) -@pytest.mark.threadleak(enabled=False) -@pytest.mark.parametrize("use_kv_cache_manager_v2", [False, True], ids=["v1", "v2"]) -def test_changed_final_token_reuse_cuda_graph( - use_kv_cache_manager_v2: bool, - monkeypatch: pytest.MonkeyPatch, -) -> None: - """Verify the promoted row reads a changed final prompt token.""" - cold, reused, graph_executions = _generate_changed_final_token_cold_and_reused( - use_kv_cache_manager_v2, - SamplingParams(max_tokens=4, end_id=-1, temperature=0), - monkeypatch, - ) - - assert PROMPT_TOKEN_IDS[:-1] == CHANGED_FINAL_PROMPT_TOKEN_IDS[:-1] - assert PROMPT_TOKEN_IDS[-1] != CHANGED_FINAL_PROMPT_TOKEN_IDS[-1] - assert cold.outputs[0].token_ids == reused.outputs[0].token_ids - _assert_reused_context_used_cuda_graph(graph_executions) - - @pytest.mark.threadleak(enabled=False) @pytest.mark.skip_less_device(2) @pytest.mark.parametrize("use_kv_cache_manager_v2", [False, True], ids=["v1", "v2"]) @@ -418,6 +399,25 @@ def test_final_token_reuse_cuda_graph_tp2( _assert_rank_reused_context_used_cuda_graph(report) +@pytest.mark.threadleak(enabled=False) +@pytest.mark.parametrize("use_kv_cache_manager_v2", [False, True], ids=["v1", "v2"]) +def test_changed_final_token_reuse_cuda_graph( + use_kv_cache_manager_v2: bool, + monkeypatch: pytest.MonkeyPatch, +) -> None: + """Verify the promoted row reads a changed final prompt token.""" + cold, reused, graph_executions = _generate_changed_final_token_cold_and_reused( + use_kv_cache_manager_v2, + SamplingParams(max_tokens=4, end_id=-1, temperature=0), + monkeypatch, + ) + + assert PROMPT_TOKEN_IDS[:-1] == CHANGED_FINAL_PROMPT_TOKEN_IDS[:-1] + assert PROMPT_TOKEN_IDS[-1] != CHANGED_FINAL_PROMPT_TOKEN_IDS[-1] + assert cold.outputs[0].token_ids == reused.outputs[0].token_ids + _assert_reused_context_used_cuda_graph(graph_executions) + + @pytest.mark.threadleak(enabled=False) @pytest.mark.parametrize("use_kv_cache_manager_v2", [False, True], ids=["v1", "v2"]) def test_context_logits_after_final_token_reuse( @@ -455,8 +455,6 @@ def test_guided_decoding_after_final_token_reuse( SamplingParams( max_tokens=4, end_id=-1, - # Keep the grammar permissive so output equality tests execution- - # path parity rather than narrow-format generation behavior. guided_decoding=GuidedDecodingParams(regex=r".*"), ), monkeypatch, diff --git a/tests/integration/defs/kv_cache/test_kv_cache_iteration_stats.py b/tests/integration/defs/kv_cache/test_kv_cache_iteration_stats.py index 6319aa85976e..e9e1c7a3b523 100644 --- a/tests/integration/defs/kv_cache/test_kv_cache_iteration_stats.py +++ b/tests/integration/defs/kv_cache/test_kv_cache_iteration_stats.py @@ -42,7 +42,7 @@ from ..conftest import llm_models_root -MODEL = f"{llm_models_root()}/llama-models-v2/TinyLlama-1.1B-Chat-v1.0" +MODEL = f"{llm_models_root()}/Qwen3/Qwen3-0.6B" ALL_FIELDS = [ # Instantaneous gauges — primary (GPU) pool diff --git a/tests/integration/defs/kv_cache/test_kv_cache_v2_scheduler.py b/tests/integration/defs/kv_cache/test_kv_cache_v2_scheduler.py index 74a03c3727b7..457b6ca7ba7d 100644 --- a/tests/integration/defs/kv_cache/test_kv_cache_v2_scheduler.py +++ b/tests/integration/defs/kv_cache/test_kv_cache_v2_scheduler.py @@ -24,9 +24,7 @@ import torch from tensorrt_llm import LLM -from tensorrt_llm.executor import request as executor_request from tensorrt_llm.llmapi import KvCacheConfig, MTPDecodingConfig, SamplingParams, SchedulerConfig -from tensorrt_llm.lora_helper import LoraConfig from ..conftest import llm_models_root, skip_pre_hopper @@ -220,276 +218,6 @@ def _run_eviction_test( return _run_v1_v2_compare(model_path, prompts, sampling_params, kv_extra=kv_extra, **llm_kwargs) -# =========================================================================== -# Functional tests on Llama-3.2-1B -# =========================================================================== -class TestKVCacheV2Llama: - """Functional tests for V2 scheduler using Llama-3.2-1B (1 GPU).""" - - MODEL_PATH = f"{llm_models_root()}/llama-3.2-models/Llama-3.2-1B" - - def _compare(self, prompts, max_tokens=32, kv_extra=None, **llm_kwargs): - """Run V1 vs V2 with greedy sampling; assert outputs match.""" - return _run_v1_v2_compare( - self.MODEL_PATH, - prompts, - SamplingParams(max_tokens=max_tokens, temperature=0.0), - kv_extra=kv_extra, - **llm_kwargs, - ) - - # Basic greedy — V2 matches V1 - def test_v2_vs_v1_basic(self): - self._compare(SHORT_PROMPTS[:5]) - - # Token budget limited — V2 matches V1 - def test_token_budget_limited(self): - self._compare(SHORT_PROMPTS, max_num_tokens=64) - - # Chunked prefill — V2 matches V1 - def test_chunked_prefill(self): - self._compare([LONG_PROMPT], max_tokens=64, enable_chunked_prefill=True, max_num_tokens=128) - - # Chunked prefill multi-request — V2 matches V1 - def test_chunked_prefill_multi_request(self): - self._compare( - MEDIUM_PROMPTS, - max_tokens=64, - # V2 can commit partial source blocks, while V1 only commits full blocks. - # Disable reuse so both managers compute the same prompt tokens for this - # output comparison. - kv_extra={"enable_block_reuse": False}, - enable_chunked_prefill=True, - max_num_tokens=256, - ) - - # Eviction — V2 matches V1 under tight memory - @pytest.mark.parametrize("use_cuda_graph", [True, False], ids=["cuda_graph", "no_cuda_graph"]) - def test_eviction(self, use_cuda_graph): - sampling_params = SamplingParams(max_tokens=64, temperature=0.0) - if use_cuda_graph: - _run_eviction_test( - self.MODEL_PATH, - EVICTION_PROMPTS_SHORT, - sampling_params, - ) - else: - # No cuda graph + tight memory: V1/V2 scheduling can diverge, - # so only assert both complete. - _run_v1_v2_compare( - self.MODEL_PATH, - SHORT_PROMPTS, - sampling_params, - kv_extra={"max_tokens": 288, "enable_block_reuse": False}, - max_batch_size=4, - max_num_tokens=256, - cuda_graph_config=None, - assert_outputs_match=False, - ) - - # Batch size limited — V2 matches V1 - def test_batch_size_limited(self): - self._compare(SHORT_PROMPTS, max_batch_size=2, max_num_tokens=8192) - - # Overlap / non-overlap scheduler — V2 matches V1 - @pytest.mark.parametrize("disable_overlap", [True, False], ids=["non_overlap", "overlap"]) - def test_overlap_scheduler(self, disable_overlap): - self._compare(SHORT_PROMPTS[:5], disable_overlap_scheduler=disable_overlap) - - # Block reuse — V2 matches V1 - def test_block_reuse(self): - self._compare(SHARED_PREFIX_PROMPTS, max_tokens=64, kv_extra={"enable_block_reuse": True}) - - # Partial block reuse — V2 matches V1 - def test_partial_block_reuse(self): - self._compare( - SHARED_PREFIX_PROMPTS, - max_tokens=64, - kv_extra={"enable_block_reuse": True, "enable_partial_reuse": True}, - ) - - # Chunked prefill + eviction — V2 matches V1 - def test_chunked_prefill_with_eviction(self): - _run_eviction_test( - self.MODEL_PATH, - [EVICTION_PROMPT] * 16, - SamplingParams(max_tokens=64, temperature=0.0), - enable_chunked_prefill=True, - max_num_tokens=256, - ) - - # Eviction + block reuse — V2 matches V1 - def test_eviction_with_block_reuse(self): - _run_eviction_test( - self.MODEL_PATH, - EVICTION_PROMPTS_SHORT, - SamplingParams(max_tokens=64, temperature=0.0), - enable_block_reuse=True, - ) - - # Chunked prefill + eviction + block reuse — V2 matches V1 - def test_chunked_prefill_eviction_block_reuse(self): - _run_eviction_test( - self.MODEL_PATH, - EVICTION_PROMPTS_SHORT, - SamplingParams(max_tokens=64, temperature=0.0), - enable_block_reuse=True, - enable_chunked_prefill=True, - max_num_tokens=256, - ) - - # Eviction + overlap scheduler — V2 matches V1 - def test_eviction_overlap(self): - _run_eviction_test( - self.MODEL_PATH, - EVICTION_PROMPTS_SHORT, - SamplingParams(max_tokens=64, temperature=0.0), - disable_overlap_scheduler=False, - ) - - -# =========================================================================== -# LoRA tests on llama-7b-hf -# =========================================================================== -@pytest.mark.skip_less_device_memory(40000) -class TestKVCacheV2LoRA: - """LoRA tests for V2 scheduler using llama-7b-hf (1 GPU, >=40GB).""" - - MODEL_PATH = f"{llm_models_root()}/llama-models/llama-7b-hf" - LORA_DIR = f"{llm_models_root()}/llama-models/luotuo-lora-7b-0.1" - LORA_CONFIG = LoraConfig( - lora_dir=[LORA_DIR], - max_lora_rank=8, - max_loras=1, - max_cpu_loras=1, - lora_target_modules=["attn_q", "attn_k", "attn_v"], - ) - - def _run_v1_v2_lora( - self, - prompts, - expected_count=None, - sampling_params=None, - kv_extra=None, - label_suffix="", - **llm_kwargs, - ): - """Run V1 then V2 with LoRA; assert outputs match.""" - if expected_count is None: - expected_count = len(prompts) - if sampling_params is None: - sampling_params = SamplingParams(max_tokens=32, temperature=0.0) - if kv_extra is None: - kv_extra = {"free_gpu_memory_fraction": 0.4} - lora_request = executor_request.LoRARequest("lora-0", 0, self.LORA_DIR) - - kv_v1 = KvCacheConfig(use_kv_cache_manager_v2=False, **kv_extra) - with LLM( - self.MODEL_PATH, - kv_cache_config=kv_v1, - lora_config=self.LORA_CONFIG, - **llm_kwargs, - ) as llm: - outputs_v1 = llm.generate( - prompts, - sampling_params=sampling_params, - lora_request=lora_request, - ) - gc.collect() - torch.cuda.empty_cache() - - kv_v2 = KvCacheConfig(use_kv_cache_manager_v2=True, **kv_extra) - with LLM( - self.MODEL_PATH, - kv_cache_config=kv_v2, - scheduler_config=_V2_SCHEDULER_CONFIG, - lora_config=self.LORA_CONFIG, - **llm_kwargs, - ) as llm: - outputs_v2 = llm.generate( - prompts, - sampling_params=sampling_params, - lora_request=lora_request, - ) - - _assert_all_completed(outputs_v1, expected_count=expected_count) - _assert_all_completed(outputs_v2, expected_count=expected_count) - _assert_outputs_match( - outputs_v1, - outputs_v2, - f"V1-LoRA{label_suffix}", - f"V2-LoRA{label_suffix}", - ) - - # Single LoRA adapter — V2 matches V1 - def test_lora_v2(self): - self._run_v1_v2_lora(SHORT_PROMPTS[:3]) - - # LoRA adapter swapping — V2 matches V1 - def test_lora_multi_adapter_v2(self): - sampling_params = SamplingParams(max_tokens=32, temperature=0.0) - lora_request = executor_request.LoRARequest("lora-0", 0, self.LORA_DIR) - - def _run_multi_adapter(kv_config, **extra_llm_kwargs): - with LLM( - self.MODEL_PATH, - kv_cache_config=kv_config, - lora_config=self.LORA_CONFIG, - **extra_llm_kwargs, - ) as llm: - out_lora = llm.generate( - SHORT_PROMPTS[:2], - sampling_params=sampling_params, - lora_request=lora_request, - ) - out_base = llm.generate( - SHORT_PROMPTS[2:4], - sampling_params=sampling_params, - ) - return out_lora, out_base - - outs_v1 = _run_multi_adapter( - KvCacheConfig(use_kv_cache_manager_v2=False, free_gpu_memory_fraction=0.4), - ) - gc.collect() - torch.cuda.empty_cache() - outs_v2 = _run_multi_adapter( - KvCacheConfig(use_kv_cache_manager_v2=True, free_gpu_memory_fraction=0.4), - scheduler_config=_V2_SCHEDULER_CONFIG, - ) - - for label, v1, v2, count in [ - ("LoRA", outs_v1[0], outs_v2[0], 2), - ("base", outs_v1[1], outs_v2[1], 2), - ]: - _assert_all_completed(v1, expected_count=count) - _assert_all_completed(v2, expected_count=count) - _assert_outputs_match(v1, v2, f"V1-{label}", f"V2-{label}") - - # LoRA + chunked prefill — V2 matches V1 - def test_lora_chunked_prefill(self): - self._run_v1_v2_lora( - MEDIUM_PROMPTS[:3], - enable_chunked_prefill=True, - max_num_tokens=128, - label_suffix="-chunked", - ) - - # LoRA + eviction — V2 matches V1 - def test_lora_eviction(self): - self._run_v1_v2_lora( - SHORT_PROMPTS, - expected_count=10, - sampling_params=SamplingParams(max_tokens=64, temperature=0.0), - kv_extra={ - "free_gpu_memory_fraction": 0.2, - "host_cache_size": 64 * 1024 * 1024, # 64 MiB host tier - }, - max_batch_size=8, - label_suffix="-evict", - ) - - # =========================================================================== # MTP tests on DeepSeek-V3-Lite (2 GPUs) # =========================================================================== diff --git a/tests/integration/defs/perf/_model_paths.py b/tests/integration/defs/perf/_model_paths.py index 1ee87c6cfd0f..b7e04c45994f 100644 --- a/tests/integration/defs/perf/_model_paths.py +++ b/tests/integration/defs/perf/_model_paths.py @@ -19,13 +19,6 @@ "llama_v3.1_8b_instruct": "llama-3.1-model/Llama-3.1-8B-Instruct", "llama_v3.1_8b_instruct_fp8": "llama-3.1-model/Llama-3.1-8B-Instruct-FP8", "llama_v3.1_8b_instruct_fp4": "modelopt-hf-model-hub/Llama-3.1-8B-Instruct-fp4", - "llama_v3.3_70b_instruct": "llama-3.3-models/Llama-3.3-70B-Instruct", - "llama_v3.3_70b_instruct_fp8": "modelopt-hf-model-hub/Llama-3.3-70B-Instruct-fp8", - "llama_v3.3_70b_instruct_fp4": "modelopt-hf-model-hub/Llama-3.3-70B-Instruct-fp4", - "llama_v3.3_nemotron_super_49b_v1.5_fp8": "nemotron-nas/Llama-3_3-Nemotron-Super-49B-v1_5-FP8", - "llama_v4_scout_17b_16e_instruct": "llama4-models/Llama-4-Scout-17B-16E-Instruct", - "llama_v4_scout_17b_16e_instruct_fp8": "llama4-models/Llama-4-Scout-17B-16E-Instruct-FP8", - "llama_v4_scout_17b_16e_instruct_fp4": "llama4-models/Llama-4-Scout-17B-16E-Instruct-FP4", "gemma_3_27b_it": "gemma/gemma-3-27b-it", "gemma_3_27b_it_fp8": "gemma/gemma-3-27b-it-fp8", "gemma_3_27b_it_fp4": "gemma/gemma-3-27b-it-FP4", diff --git a/tests/integration/defs/perf/pytorch_model_config.py b/tests/integration/defs/perf/pytorch_model_config.py index 98d6fd11319d..2d614aee96a1 100644 --- a/tests/integration/defs/perf/pytorch_model_config.py +++ b/tests/integration/defs/perf/pytorch_model_config.py @@ -375,20 +375,6 @@ def get_model_yaml_config(model_label: str, }, } }, - # Llama-v4 Scout FP4 with cuda graph padding - { - 'patterns': ['llama_v4_scout_17b_16e_instruct_fp4'], - 'config': { - 'cuda_graph_config': { - 'enable_padding': - True, - 'batch_sizes': [ - 1, 2, 4, 8, 16, 32, 64, 128, 256, 384, 512, 1024, 2048, - 4096, 8192 - ] - } - } - }, # GPT-OSS 120B max throughput test { 'patterns': [ diff --git a/tests/integration/defs/perf/sampler_options_config.py b/tests/integration/defs/perf/sampler_options_config.py index de1824c8eced..f806d5b6504b 100644 --- a/tests/integration/defs/perf/sampler_options_config.py +++ b/tests/integration/defs/perf/sampler_options_config.py @@ -30,10 +30,4 @@ def get_sampler_options_config(model_label: str) -> dict: # PerfTestConfig.to_string() emits them: maxbs:/maxnt: are always injected # and tp: is dropped when tp_size == num_gpus. base_config = {} - if model_label in [ - 'llama_v3.3_70b_instruct_fp8-bench-pytorch-float8-maxbs:512-maxnt:2048-input_output_len:128,128-gpus:8', - ]: - base_config['top_k'] = 4 - base_config['top_p'] = 0.5 - base_config['temperature'] = 0.5 return base_config diff --git a/tests/integration/defs/test_e2e.py b/tests/integration/defs/test_e2e.py index 04df11cc07eb..e1353abb6816 100644 --- a/tests/integration/defs/test_e2e.py +++ b/tests/integration/defs/test_e2e.py @@ -12,7 +12,6 @@ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. # See the License for the specific language governing permissions and # limitations under the License. -import json import os import re import shutil @@ -233,75 +232,6 @@ def parse_benchmark_output(self, output): return result -@pytest.mark.parametrize("model_name", ["meta-llama/Meta-Llama-3-8B-Instruct"], - ids=["llama3-8b"]) -@pytest.mark.parametrize("model_subdir", - ["llama-models-v3/llama-v3-8b-instruct-hf"], - ids=["llama-v3"]) -@pytest.mark.parametrize("use_pytorch_backend", [True], ids=["pytorch_backend"]) -def test_trtllm_bench_llmapi_launch(llm_root, llm_venv, model_name, - model_subdir, use_pytorch_backend): - runner = BenchRunner(llm_root=llm_root, - llm_venv=llm_venv, - model_name=model_name, - model_subdir=model_subdir, - streaming=False, - use_mpirun=True, - tp_size=2) - runner() - - -@pytest.mark.parametrize( - "model_name, llama_model_root", - [pytest.param("TinyLlama-1.1B-Chat-v1.0", "TinyLlama-1.1B-Chat-v1.0")], - indirect=["llama_model_root"]) -def test_trtllm_bench_invalid_token_pytorch(llm_root, llm_venv, model_name, - llama_model_root): - # Prepare dataset with invalid tokens - _, dataset_path = trtllm_bench_prolog(llm_root, - llm_venv, - model_subdir=llama_model_root, - model_name=model_name, - quant=None, - streaming=False) - with open(dataset_path) as f: - dataset = [json.loads(line) for line in f.readlines()] - dataset[0]["input_ids"][-1] = -1 - with open(dataset_path, "w") as f: - f.writelines(f"{json.dumps(data)}\n" for data in dataset) - - # Run benchmark - extra_options = { - "cuda_graph_config": { - "enable_padding": True, - "batch_sizes": [1, 2, 4, 8, 16, 32, 64, 128, 256, 384], - }, - } - with tempfile.TemporaryDirectory() as tmpdir: - extra_options_path = Path(tmpdir) / "extra-llm-api-options.yml" - with open(extra_options_path, "w") as f: - yaml.dump(extra_options, f) - - output_path = Path(tmpdir) / "stdout.log" - benchmark_cmd = \ - f"trtllm-bench --model {model_name} " \ - f"--model_path {llama_model_root} " \ - f"throughput " \ - f"--dataset {str(dataset_path)} --backend pytorch " \ - f"--config {extra_options_path} " \ - f"> {output_path} 2>&1" - # Check clean shutdown (no hang) - with pytest.raises(subprocess.CalledProcessError) as exc_info: - check_call(benchmark_cmd, shell=True, env=llm_venv._new_env) - # Check non-zero exit code - assert exc_info.value.returncode != 0 - with open(output_path) as f: - stdout = f.read() - - # Check that error is reported correctly - assert "Requests failed: Token ID out of range (1 requests)" in stdout - - def trtllm_bench_prolog(llm_root, llm_venv, model_subdir, model_name: str, quant: str, streaming: bool) -> Tuple[Path, Path]: """Generate dataset for benchmark. @@ -878,7 +808,6 @@ def test_ptp_quickstart(llm_root, llm_venv): @pytest.mark.parametrize("model_name,model_path", [ ("Llama3.1-8B-BF16", "llama-3.1-model/Meta-Llama-3.1-8B"), - ("Llama3.2-11B-BF16", "llama-3.2-models/Llama-3.2-11B-Vision"), ("Nemotron4_4B-BF16", "nemotron/Minitron-4B-Base"), pytest.param('Llama3.1-8B-NVFP4', 'nvfp4-quantized/Meta-Llama-3.1-8B', @@ -886,12 +815,6 @@ def test_ptp_quickstart(llm_root, llm_venv): pytest.param('Llama3.1-8B-FP8', 'llama-3.1-model/Llama-3.1-8B-Instruct-FP8', marks=skip_pre_hopper), - pytest.param('Nemotron-Super-49B-v1-NVFP4', - 'nvfp4-quantized/Llama-3_3-Nemotron-Super-49B-v1_nvfp4_hf', - marks=skip_pre_hopper), - pytest.param('Nemotron-Super-49B-v1-FP8', - 'nemotron-nas/Llama-3_3-Nemotron-Super-49B-v1-FP8', - marks=skip_pre_hopper), pytest.param('Mixtral-8x7B-NVFP4', 'nvfp4-quantized/Mixtral-8x7B-Instruct-v0.1', marks=skip_pre_blackwell), @@ -909,16 +832,6 @@ def test_ptp_quickstart(llm_root, llm_venv): 'Qwen3-30B-A3B_nvfp4_hf', 'Qwen3/saved_models_Qwen3-30B-A3B_nvfp4_hf', marks=(skip_pre_blackwell, pytest.mark.skip_less_device_memory(20000))), - pytest.param( - 'Llama3.3-70B-FP8', - 'modelopt-hf-model-hub/Llama-3.3-70B-Instruct-fp8', - marks=(skip_pre_blackwell, pytest.mark.skip_less_device_memory(96000))), - pytest.param('Llama3.3-70B-FP4', - 'modelopt-hf-model-hub/Llama-3.3-70B-Instruct-fp4', - marks=skip_pre_blackwell), - pytest.param('Nemotron-Super-49B-v1-BF16', - 'nemotron-nas/Llama-3_3-Nemotron-Super-49B-v1', - marks=skip_pre_blackwell), pytest.param('Mixtral-8x7B-BF16', 'Mixtral-8x7B-Instruct-v0.1', marks=skip_pre_blackwell), @@ -956,12 +869,6 @@ def test_ptp_quickstart(llm_root, llm_venv): 'nvidia-Phi-4-reasoning-plus-NVFP4', marks=skip_pre_blackwell), ("Phi-4-reasoning-plus-bf16", "Phi-4-reasoning-plus"), - pytest.param('Nemotron-Super-49B-v1.5-FP8', - 'nemotron-nas/Llama-3_3-Nemotron-Super-49B-v1_5-FP8', - marks=skip_pre_hopper), - pytest.param('Llama-4-Scout-17B-16E-FP4', - 'llama4-models/Llama-4-Scout-17B-16E-Instruct-FP4', - marks=skip_pre_blackwell), pytest.param('Nemotron-Nano-9B-v2-nvfp4', 'NVIDIA-Nemotron-Nano-9B-v2-NVFP4', marks=skip_pre_blackwell), @@ -980,7 +887,6 @@ def test_ptp_quickstart_advanced(llm_root, llm_venv, model_name, model_path): else: mapping = { "Llama3.1-8B-BF16": 18.60, - "Llama3.2-11B-BF16": 18.88, "Nemotron4_4B-BF16": 12.50, "Llama3.1-8B-FP8": 13.05, "Llama3.1-8B-NVFP4": 10.2 @@ -992,8 +898,6 @@ def test_ptp_quickstart_advanced(llm_root, llm_venv, model_name, model_path): ] if "Qwen3" in model_name: cmds.append("--kv_cache_fraction=0.6") - if "Llama3.1-70B" in model_name or "Llama3.3-70B" in model_name: - cmds.append("--max_num_tokens=1024") llm_venv.run_cmd(cmds) @@ -1418,10 +1322,6 @@ def test_deepseek_r1_mtp_bench(llm_root, llm_venv): 'nvfp4-quantized/Mixtral-8x7B-Instruct-v0.1', 8, marks=skip_pre_blackwell), - pytest.param('Nemotron-Ultra-253B', - 'nemotron-nas/Llama-3_1-Nemotron-Ultra-253B-v1', - 8, - marks=(skip_pre_hopper, pytest.mark.timeout(12600))), pytest.param('DeepSeek-V3-671B-FP8', 'DeepSeek-V3-0324', 8, @@ -1440,7 +1340,6 @@ def test_ptp_quickstart_advanced_multi_gpus(llm_root, llm_venv, model_name, "Llama3.1-70B-FP8": 58.5, "Llama3.1-405B-FP8": 63.2, "Mixtral-8x7B-NVFP4": 9.9, - "Nemotron-Ultra-253B": 72.3, "DeepSeek-V3-671B-FP8": 83.8 } llm_venv.run_cmd([ @@ -1454,82 +1353,12 @@ def test_ptp_quickstart_advanced_multi_gpus(llm_root, llm_venv, model_name, ]) -@pytest.mark.skip_less_device_memory(80000) -@pytest.mark.parametrize("cuda_graph", [False, True]) -@pytest.mark.parametrize("tp_size, pp_size", [ - pytest.param(2, 2, marks=pytest.mark.skip_less_device(4)), - pytest.param(2, 4, marks=pytest.mark.skip_less_mpi_world_size(8)), -]) -@pytest.mark.parametrize("model_name,model_path", [ - pytest.param('Llama3.3-70B-FP8', - 'llama-3.3-models/Llama-3.3-70B-Instruct-FP8', - marks=skip_pre_hopper), -]) -def test_ptp_quickstart_advanced_pp_enabled(llm_root, llm_venv, model_name, - model_path, cuda_graph, tp_size, - pp_size): - print(f"Testing {model_name} on 8 GPUs.") - example_root = Path(os.path.join(llm_root, "examples", "llm-api")) - cmd = [ - str(example_root / "quickstart_advanced.py"), - "--enable_chunked_prefill", - "--model_dir", - f"{llm_models_root()}/{model_path}", - f"--tp_size={tp_size}", - f"--pp_size={pp_size}", - "--moe_ep_size=1", - "--kv_cache_fraction=0.6", - ] - if cuda_graph: - cmd.extend([ - "--use_cuda_graph", - "--cuda_graph_padding_enabled", - ]) - llm_venv.run_cmd(cmd) - - -@skip_pre_hopper -@pytest.mark.skip_less_mpi_world_size(8) -@pytest.mark.parametrize("cuda_graph", [False, True]) -@pytest.mark.parametrize("model_name,model_path", [ - ("Llama-4-Maverick-17B-128E-Instruct-FP8", - "llama4-models/nvidia/Llama-4-Maverick-17B-128E-Instruct-FP8"), - ("Llama-4-Scout-17B-16E-Instruct-FP8", - "llama4-models/Llama-4-Scout-17B-16E-Instruct-FP8"), - pytest.param('Llama-4-Scout-17B-16E-Instruct-FP4', - 'llama4-models/Llama-4-Scout-17B-16E-Instruct-FP4', - marks=skip_pre_blackwell), -]) -def test_ptp_quickstart_advanced_8gpus_chunked_prefill_sq_22k( - llm_root, llm_venv, model_name, model_path, cuda_graph): - print(f"Testing {model_name} on 8 GPUs.") - example_root = Path(os.path.join(llm_root, "examples", "llm-api")) - cmd = [ - str(example_root / "quickstart_advanced.py"), - "--enable_chunked_prefill", - "--model_dir", - f"{llm_models_root()}/{model_path}", - "--tp_size=8", - "--moe_ep_size=8", - "--max_seq_len=22000", - "--kv_cache_fraction=0.6", - ] - if cuda_graph: - cmd.extend([ - "--use_cuda_graph", - "--cuda_graph_padding_enabled", - ]) - llm_venv.run_cmd(cmd) - - # This test is specifically to be run on 2 GPUs on Blackwell RTX 6000 Pro (SM120) architecture # TODO: remove once we have a node with 8 GPUs and reuse test_ptp_quickstart_advanced_8gpus @skip_no_sm120 @pytest.mark.skip_less_device_memory(80000) @pytest.mark.skip_less_device(2) @pytest.mark.parametrize("model_name,model_path", [ - ('Nemotron-Super-49B-v1-BF16', - 'nemotron-nas/Llama-3_3-Nemotron-Super-49B-v1'), ("Mixtral-8x7B-BF16", "Mixtral-8x7B-Instruct-v0.1"), ]) def test_ptp_quickstart_advanced_2gpus_sm120(llm_root, llm_venv, model_name, @@ -1916,14 +1745,8 @@ def test_multi_nodes_eval(model_path, tp_size, pp_size, ep_size, eval_task, @pytest.mark.parametrize("tp_size,pp_size", [(2, 1), (1, 2)], ids=["tp2", "pp2"]) @pytest.mark.parametrize("model_path", [ - pytest.param('llama-3.3-models/Llama-3.3-70B-Instruct', - marks=skip_pre_hopper), pytest.param('Qwen3/saved_models_Qwen3-235B-A22B_nvfp4_hf', marks=skip_pre_blackwell), - pytest.param('llama4-models/Llama-4-Scout-17B-16E-Instruct-FP8', - marks=skip_pre_hopper), - pytest.param('llama4-models/Llama-4-Scout-17B-16E-Instruct', - marks=skip_pre_hopper), ]) def test_ptp_quickstart_advanced_multinode(llm_root, llm_venv, model_path, tp_size, pp_size): @@ -1937,8 +1760,7 @@ def test_ptp_quickstart_advanced_multinode(llm_root, llm_venv, model_path, model_dir = f"{llm_models_root()}/{model_path}" prompt = "Explain why New York is great city to live in, in 1 short paragraph" - moe_ep_size = tp_size if ("Llama-4" in model_path - or "Qwen3" in model_path) and tp_size > 1 else -1 + moe_ep_size = tp_size if "Qwen3" in model_path and tp_size > 1 else -1 with LLM( model=model_dir, @@ -1964,15 +1786,10 @@ def test_ptp_quickstart_advanced_multinode(llm_root, llm_venv, model_path, @pytest.mark.skip_less_device_memory(80000) @skip_pre_hopper @pytest.mark.parametrize("model_dir,draft_model_dir", [ - ("modelopt-hf-model-hub/Llama-3.3-70B-Instruct-fp8", - "EAGLE3-LLaMA3.3-Instruct-70B"), ("Qwen3/Qwen3-30B-A3B", "Qwen3/Qwen3-30B-eagle3"), pytest.param("Qwen3/saved_models_Qwen3-235B-A22B_fp8_hf", "Qwen3/qwen3-235B-eagle3", marks=pytest.mark.skip_less_device_memory(90000)), - pytest.param("llama4-models/nvidia/Llama-4-Maverick-17B-128E-Instruct-FP8", - "Llama-4-Maverick-17B-128E-Eagle3", - marks=pytest.mark.skip_less_device_memory(140000)), pytest.param("Qwen3/saved_models_Qwen3-235B-A22B_nvfp4_hf", "Qwen3/qwen3-235B-eagle3", marks=skip_pre_blackwell), diff --git a/tests/integration/test_lists/qa/llm_function_core.txt b/tests/integration/test_lists/qa/llm_function_core.txt index 5c48a75d648d..347c2bafca1a 100644 --- a/tests/integration/test_lists/qa/llm_function_core.txt +++ b/tests/integration/test_lists/qa/llm_function_core.txt @@ -85,7 +85,6 @@ accuracy/test_llm_api_autodeploy.py::TestModelRegistryAccuracy::test_autodeploy_ accuracy/test_llm_api_autodeploy.py::TestModelRegistryAccuracy::test_autodeploy_from_registry[deepseek-ai_DeepSeek-R1-0528-True] accuracy/test_llm_api_autodeploy.py::TestModelRegistryAccuracy::test_autodeploy_from_registry[google_gemma-3-1b-it-False] accuracy/test_llm_api_autodeploy.py::TestModelRegistryAccuracy::test_autodeploy_from_registry[meta-llama_Llama-3.1-8B-Instruct-False] -accuracy/test_llm_api_autodeploy.py::TestModelRegistryAccuracy::test_autodeploy_from_registry[meta-llama_Llama-3.3-70B-Instruct-False] accuracy/test_llm_api_autodeploy.py::TestModelRegistryAccuracy::test_autodeploy_from_registry[mistralai_Codestral-22B-v0.1-False] accuracy/test_llm_api_autodeploy.py::TestModelRegistryAccuracy::test_autodeploy_from_registry[mistralai_Ministral-8B-Instruct-2410-False] accuracy/test_llm_api_autodeploy.py::TestModelRegistryAccuracy::test_autodeploy_from_registry[nvidia_DeepSeek-R1-0528-NVFP4-v2-True] @@ -626,20 +625,6 @@ accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_pard_sa accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_pard_sa_dynamic_draft_len accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_pard_sa_global_pool accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_suffix_automaton_dynamic_draft_len -accuracy/test_llm_api_pytorch.py::TestLlama3_1_8B_Instruct_RocketKV::test_auto_dtype -accuracy/test_llm_api_pytorch.py::TestLlama3_2_3B::test_auto_dtype -accuracy/test_llm_api_pytorch.py::TestLlama3_2_3B::test_fp8_prequantized -accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp4_tp2pp2[torch_compile=False-enable_gemm_allreduce_fusion=False] -accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp4_tp2pp2[torch_compile=False-enable_gemm_allreduce_fusion=True] -accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp4_tp2pp2[torch_compile=True-enable_gemm_allreduce_fusion=False] -accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp4_tp2pp2[torch_compile=True-enable_gemm_allreduce_fusion=True] -accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp8_eagle3_tp8[eagle3_one_model=False-torch_compile=False] -accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp8_eagle3_tp8[eagle3_one_model=True-torch_compile=False] -accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp8_eagle3_tp8[eagle3_one_model=True-torch_compile=True] -accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp8_tp4[torch_compile=False] -accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp8_tp4[torch_compile=True] -accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_nvfp4_tp4[torch_compile=False] -accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_nvfp4_tp4[torch_compile=True] accuracy/test_llm_api_pytorch.py::TestMiniMaxM2::test_4gpus[attention_dp=False-cuda_graph=True-overlap_scheduler=True-tp_size=4-ep_size=4] accuracy/test_llm_api_pytorch.py::TestMiniMaxM2_5::test_4gpus[attention_dp=False-cuda_graph=True-overlap_scheduler=True-tp_size=4-ep_size=4] accuracy/test_llm_api_pytorch.py::TestMiniMaxM3::test_auto_dtype[tp_size=8-ep_size=8] TIMEOUT (180) @@ -776,7 +761,12 @@ accuracy/test_llm_api_pytorch.py::TestStep3_7::test_fp8_block_scales[tp_size=4-e accuracy/test_llm_api_pytorch.py::TestStep3_7::test_fp8_block_scales[tp_size=4-ep_size=4-mtp_nextn=3] TIMEOUT (90) accuracy/test_llm_api_pytorch.py::TestStep3_7::test_nvfp4[tp_size=4-ep_size=4-mtp_nextn=0] TIMEOUT (90) accuracy/test_llm_api_pytorch.py::TestStep3_7::test_nvfp4[tp_size=4-ep_size=4-mtp_nextn=3] TIMEOUT (90) -accuracy/test_llm_api_pytorch_encode.py::TestDecoderEncode::test_decoder_encode_cuda_graph_matches_eager_logits[tinyllama-1.1b] +accuracy/test_llm_api_pytorch_encode.py::TestEncoderEncode::test_encoder_encode_matches_huggingface_classification[bert-yelp-eager] +accuracy/test_llm_api_pytorch_encode.py::TestEncoderEncode::test_encoder_encode_matches_huggingface_classification[bert-yelp-cuda_graph] +accuracy/test_llm_api_pytorch_encode.py::TestEncoderEncode::test_encoder_encode_cuda_graph_matches_eager_logits[bert-yelp] +accuracy/test_llm_api_pytorch_encode.py::TestEncoderEncode::test_encoder_encode_matches_huggingface_per_token_reward[qwen2.5-prm-7b] +accuracy/test_llm_api_pytorch_encode.py::TestEncoderEncode::test_qwen3_text_embedding_matches_huggingface[qwen3-embedding-0.6b] +accuracy/test_llm_api_pytorch_encode.py::TestEncoderEncode::test_qwen3_text_embedding_matches_huggingface[qwen3-embedding-8b] accuracy/test_llm_api_pytorch_encode.py::TestDecoderEncode::test_decoder_encode_matches_huggingface[gemma-3-1b] accuracy/test_llm_api_pytorch_encode.py::TestDecoderEncode::test_decoder_encode_matches_huggingface[phi-4-mini] accuracy/test_llm_api_pytorch_encode.py::TestDecoderEncode::test_decoder_encode_matches_huggingface[qwen2-7b] @@ -816,7 +806,10 @@ disaggregated/test_ad_disagg.py::test_async_eagle3_full_model_handoff disaggregated/test_ad_disagg.py::test_async_generation_matches_aggregate disaggregated/test_ad_disagg.py::test_async_generation_no_overlap_matches_aggregate disaggregated/test_ad_disagg.py::test_async_sharded_generation_handoff -disaggregated/test_ad_disagg_trtllm_serve.py::test_openai_completion +disaggregated/test_aiperf_gate.py::test_fires_on_error_storm +disaggregated/test_aiperf_gate.py::test_passes_healthy_run_with_cancellations +disaggregated/test_aiperf_gate.py::test_missing_export_raises +disaggregated/test_aiperf_gate.py::test_empty_export_fails disaggregated/test_aiperf_gate.py::test_all_cancelled_fails disaggregated/test_aiperf_gate.py::test_corrupt_export_fails disaggregated/test_aiperf_gate.py::test_empty_export_fails @@ -844,18 +837,15 @@ disaggregated/test_auto_scaling.py::test_worker_restart[etcd-round_robin] disaggregated/test_auto_scaling.py::test_worker_restart[http-kv_cache_aware] disaggregated/test_auto_scaling.py::test_worker_restart[http-load_balancing] disaggregated/test_auto_scaling.py::test_worker_restart[http-round_robin] -disaggregated/test_disaggregated.py::test_disaggregated_benchmark_gen_only_insufficient_kv[TinyLlama-1.1B-Chat-v1.0] -disaggregated/test_disaggregated.py::test_disaggregated_cache_aware_balance[TinyLlama-1.1B-Chat-v1.0] +disaggregated/test_disaggregated.py::test_disaggregated_benchmark_gen_only +disaggregated/test_disaggregated.py::test_disaggregated_benchmark_gen_only_insufficient_kv disaggregated/test_disaggregated.py::test_disaggregated_cancel_large_context_requests[DeepSeek-V3-Lite-bf16] -disaggregated/test_disaggregated.py::test_disaggregated_chat_completion_tool_calls[TinyLlama-1.1B-Chat-v1.0] -disaggregated/test_disaggregated.py::test_disaggregated_conditional[TinyLlama-1.1B-Chat-v1.0] -disaggregated/test_disaggregated.py::test_disaggregated_ctxpp2_genpp2[TinyLlama-1.1B-Chat-v1.0] -disaggregated/test_disaggregated.py::test_disaggregated_ctxpp2_gentp2[TinyLlama-1.1B-Chat-v1.0] -disaggregated/test_disaggregated.py::test_disaggregated_ctxpp4_genpp4[TinyLlama-1.1B-Chat-v1.0] -disaggregated/test_disaggregated.py::test_disaggregated_ctxpp4_gentp4[TinyLlama-1.1B-Chat-v1.0] -disaggregated/test_disaggregated.py::test_disaggregated_ctxtp2_genpp2[TinyLlama-1.1B-Chat-v1.0] -disaggregated/test_disaggregated.py::test_disaggregated_ctxtp2pp2_gentp2pp2[TinyLlama-1.1B-Chat-v1.0] -disaggregated/test_disaggregated.py::test_disaggregated_cuda_graph[TinyLlama-1.1B-Chat-v1.0] +disaggregated/test_disaggregated.py::test_disaggregated_ctxpp2_genpp2 +disaggregated/test_disaggregated.py::test_disaggregated_ctxpp4_genpp4 +disaggregated/test_disaggregated.py::test_disaggregated_ctxpp4_gentp4 +disaggregated/test_disaggregated.py::test_disaggregated_ctxtp2_genpp2 +disaggregated/test_disaggregated.py::test_disaggregated_ctxtp2pp2_gentp2pp2 +disaggregated/test_disaggregated.py::test_disaggregated_cuda_graph disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_bf16_cache_aware_balance[DeepSeek-V3-Lite-bf16] disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_bf16_conditional[DeepSeek-V3-Lite-bf16] disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_bf16_conditional_v2[DeepSeek-V3-Lite-bf16] @@ -880,48 +870,31 @@ disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_tp1 disaggregated/test_disaggregated.py::test_disaggregated_diff_max_tokens[TinyLlama-1.1B-Chat-v1.0] disaggregated/test_disaggregated.py::test_disaggregated_genbs1[TinyLlama-1.1B-Chat-v1.0] disaggregated/test_disaggregated.py::test_disaggregated_gpt_oss_120b_harmony[gpt_oss/gpt-oss-120b] -disaggregated/test_disaggregated.py::test_disaggregated_kv_cache_time_output[TinyLlama-1.1B-Chat-v1.0] -disaggregated/test_disaggregated.py::test_disaggregated_load_balance[TinyLlama-1.1B-Chat-v1.0] +disaggregated/test_disaggregated.py::test_disaggregated_kv_cache_time_output +disaggregated/test_disaggregated.py::test_disaggregated_load_balance disaggregated/test_disaggregated.py::test_disaggregated_logprobs_serving[llama-3.1-8b-instruct] disaggregated/test_disaggregated.py::test_disaggregated_mamba_bs1_concurrency2 -disaggregated/test_disaggregated.py::test_disaggregated_mixed[TinyLlama-1.1B-Chat-v1.0] -disaggregated/test_disaggregated.py::test_disaggregated_multi_gpu[TinyLlama-1.1B-Chat-v1.0] -disaggregated/test_disaggregated.py::test_disaggregated_ngram[TinyLlama-1.1B-Chat-v1.0] -disaggregated/test_disaggregated.py::test_disaggregated_overlap[TinyLlama-1.1B-Chat-v1.0] -disaggregated/test_disaggregated.py::test_disaggregated_overlap_gen_first[ctx_pp1-TinyLlama-1.1B-Chat-v1.0] -disaggregated/test_disaggregated.py::test_disaggregated_overlap_gen_first[ctx_pp4-TinyLlama-1.1B-Chat-v1.0] -disaggregated/test_disaggregated.py::test_disaggregated_overlap_transceiver_runtime_python[TinyLlama-1.1B-Chat-v1.0] -disaggregated/test_disaggregated.py::test_disaggregated_perf_metrics[TinyLlama-1.1B-Chat-v1.0] -disaggregated/test_disaggregated.py::test_disaggregated_single_gpu[TinyLlama-1.1B-Chat-v1.0] -disaggregated/test_disaggregated_single_gpu.py::test_arbitrary_kv_cache_transfer[False-TinyLlama-1.1B-Chat-v1.0] -disaggregated/test_disaggregated_single_gpu.py::test_arbitrary_kv_cache_transfer_missing_blocks[False-TinyLlama-1.1B-Chat-v1.0] -disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_cancel_gen_requests[TinyLlama-1.1B-Chat-v1.0] +disaggregated/test_disaggregated.py::test_disaggregated_mixed +disaggregated/test_disaggregated.py::test_disaggregated_multi_orchestrator +disaggregated/test_disaggregated.py::test_disaggregated_ngram +disaggregated/test_disaggregated.py::test_disaggregated_overlap_gen_first[ctx_pp1] +disaggregated/test_disaggregated.py::test_disaggregated_overlap_gen_first[ctx_pp4] +disaggregated/test_disaggregated.py::test_disaggregated_overlap_transceiver_runtime_python +disaggregated/test_disaggregated.py::test_disaggregated_perf_metrics +disaggregated/test_disaggregated.py::test_disaggregated_python_transceiver_host_offload disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_llama_context_capacity[False-False-DeepSeek-V3-Lite-fp8/fp8] -disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_logits[False-TinyLlama-1.1B-Chat-v1.0] -disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_logits[True-TinyLlama-1.1B-Chat-v1.0] -disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_logprobs[False-TinyLlama-1.1B-Chat-v1.0] -disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_logprobs[True-TinyLlama-1.1B-Chat-v1.0] disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_simple_deepseek[False-False-DeepSeek-V3-Lite-fp8/fp8] disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_simple_deepseek[False-True-DeepSeek-V3-Lite-fp8/fp8] disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_simple_deepseek[True-False-DeepSeek-V3-Lite-fp8/fp8] disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_simple_deepseek[True-True-DeepSeek-V3-Lite-fp8/fp8] -disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_simple_llama[False-False-TinyLlama-1.1B-Chat-v1.0] -disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_simple_llama[False-True-TinyLlama-1.1B-Chat-v1.0] -disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_simple_llama[True-False-TinyLlama-1.1B-Chat-v1.0] -disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_simple_llama[True-True-TinyLlama-1.1B-Chat-v1.0] disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_simple_qwen3[False-False-Qwen3-8B-FP8] disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_simple_qwen3[False-True-Qwen3-8B-FP8] disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_simple_qwen3[True-False-Qwen3-8B-FP8] disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_simple_qwen3[True-True-Qwen3-8B-FP8] disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_spec_dec_batch_slot_limit[False-False-EAGLE3-LLaMA3.1-Instruct-8B-Llama-3.1-8B-Instruct] disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_spec_dec_batch_slot_limit[True-False-EAGLE3-LLaMA3.1-Instruct-8B-Llama-3.1-8B-Instruct] -disaggregated/test_workers.py::test_workers_conditional_disaggregation[TinyLlama-1.1B-Chat-v1.0] disaggregated/test_workers.py::test_workers_conditional_disaggregation_deepseek_v3_lite_bf16[DeepSeek-V3-Lite-bf16] -disaggregated/test_workers.py::test_workers_conversation_router[TinyLlama-1.1B-Chat-v1.0] -disaggregated/test_workers.py::test_workers_kv_cache_aware_router[TinyLlama-1.1B-Chat-v1.0] disaggregated/test_workers.py::test_workers_kv_cache_aware_router_deepseek_v3_lite_bf16[DeepSeek-V3-Lite-bf16] -disaggregated/test_workers.py::test_workers_kv_cache_aware_router_eviction[TinyLlama-1.1B-Chat-v1.0] -disaggregated/test_workers.py::test_workers_kv_cache_events[TinyLlama-1.1B-Chat-v1.0] kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix_smoke llmapi/test_llm_api_pytorch_moe_lora.py::test_mixtral_moe_routed_expert_fp8_multi_lora_varying_ranks[cudagraph] TIMEOUT (90) llmapi/test_llm_api_pytorch_moe_lora.py::test_mixtral_moe_routed_expert_fp8_multi_lora_varying_ranks[eager] TIMEOUT (90) @@ -933,20 +906,43 @@ llmapi/test_llm_examples.py::test_llmapi_server_example test_e2e.py::test_eagle3_output_repetition_4gpus[Qwen3/Qwen3-30B-A3B-Qwen3/Qwen3-30B-eagle3] test_e2e.py::test_eagle3_output_repetition_4gpus[Qwen3/saved_models_Qwen3-235B-A22B_fp8_hf-Qwen3/qwen3-235B-eagle3] test_e2e.py::test_eagle3_output_repetition_4gpus[Qwen3/saved_models_Qwen3-235B-A22B_nvfp4_hf-Qwen3/qwen3-235B-eagle3] -test_e2e.py::test_eagle3_output_repetition_4gpus[modelopt-hf-model-hub/Llama-3.3-70B-Instruct-fp8-EAGLE3-LLaMA3.3-Instruct-70B] test_e2e.py::test_openai_chat_guided_decoding[openai/gpt-oss-120b] test_e2e.py::test_openai_chat_harmony_perf_metrics test_e2e.py::test_openai_kv_cache_contamination test_e2e.py::test_ptp_quickstart_advanced[Llama3.1-8B-BF16-llama-3.1-model/Meta-Llama-3.1-8B] test_e2e.py::test_ptp_quickstart_advanced[Qwen3-30B-A3B-Qwen3/Qwen3-30B-A3B] test_e2e.py::test_ptp_quickstart_advanced_deepseek_r1_8gpus[DeepSeek-R1-DeepSeek-R1/DeepSeek-R1] +disaggregated/test_auto_scaling.py::test_service_discovery[etcd-round_robin] +disaggregated/test_auto_scaling.py::test_service_discovery[etcd-load_balancing] +disaggregated/test_auto_scaling.py::test_service_discovery[etcd-kv_cache_aware] +disaggregated/test_auto_scaling.py::test_service_discovery[http-round_robin] +disaggregated/test_auto_scaling.py::test_service_discovery[http-load_balancing] +disaggregated/test_auto_scaling.py::test_service_discovery[http-kv_cache_aware] +disaggregated/test_auto_scaling.py::test_minimal_instances[etcd-round_robin] +disaggregated/test_auto_scaling.py::test_minimal_instances[http-round_robin] +disaggregated/test_auto_scaling.py::test_worker_restart[etcd-round_robin] +disaggregated/test_auto_scaling.py::test_worker_restart[etcd-load_balancing] +disaggregated/test_auto_scaling.py::test_worker_restart[etcd-kv_cache_aware] +disaggregated/test_auto_scaling.py::test_worker_restart[http-round_robin] +disaggregated/test_auto_scaling.py::test_worker_restart[http-load_balancing] +disaggregated/test_auto_scaling.py::test_worker_restart[http-kv_cache_aware] +disaggregated/test_auto_scaling.py::test_disagg_server_restart[etcd-round_robin] +disaggregated/test_auto_scaling.py::test_disagg_server_restart[http-round_robin] +disaggregated/test_disaggregated.py::test_disaggregated_cache_aware_balance[Qwen3-0.6B] +disaggregated/test_disaggregated.py::test_disaggregated_chat_completion_tool_calls[Qwen3-0.6B] +disaggregated/test_disaggregated.py::test_disaggregated_conditional[Qwen3-0.6B] +disaggregated/test_disaggregated.py::test_disaggregated_overlap[Qwen3-0.6B] +disaggregated/test_disaggregated.py::test_disaggregated_single_gpu[Qwen3-0.6B] +disaggregated/test_disaggregated_single_gpu.py::test_arbitrary_kv_cache_transfer[False-Qwen3-0.6B] +disaggregated/test_disaggregated_single_gpu.py::test_arbitrary_kv_cache_transfer_missing_blocks[False-Qwen3-0.6B] +disaggregated/test_workers.py::test_workers_conditional_disaggregation[Qwen3-0.6B] +disaggregated/test_workers.py::test_workers_kv_cache_aware_router[Qwen3-0.6B] +disaggregated/test_workers.py::test_workers_kv_cache_aware_router_eviction[Qwen3-0.6B] +disaggregated/test_workers.py::test_workers_kv_cache_events[Qwen3-0.6B] +test_e2e.py::test_ptp_quickstart_advanced[Llama3.1-8B-BF16-llama-3.1-model/Meta-Llama-3.1-8B] test_e2e.py::test_ptp_quickstart_advanced_deepseek_r1_w4afp8_8gpus[DeepSeek-R1-W4AFP8-DeepSeek-R1/DeepSeek-R1-W4AFP8] -test_e2e.py::test_ptp_quickstart_advanced_pp_enabled[Llama3.3-70B-FP8-llama-3.3-models/Llama-3.3-70B-Instruct-FP8-2-2-False] -test_e2e.py::test_ptp_quickstart_advanced_pp_enabled[Llama3.3-70B-FP8-llama-3.3-models/Llama-3.3-70B-Instruct-FP8-2-4-True] test_e2e.py::test_ptp_quickstart_bert[TRTLLM-BertForSequenceClassification-bert/bert-base-uncased-yelp-polarity] test_e2e.py::test_ptp_quickstart_bert[VANILLA-BertForSequenceClassification-bert/bert-base-uncased-yelp-polarity] -test_e2e.py::test_qwen_e2e_cpprunner_large_new_tokens[Qwen3/Qwen3-0.6B-Qwen3/Qwen3-0.6B] test_e2e.py::test_relaxed_acceptance_quickstart_advanced_deepseek_r1_8gpus[DeepSeek-R1-DeepSeek-R1/DeepSeek-R1] test_e2e.py::test_trtllm_benchmark_serving[gpt_oss/gpt-oss-20b] test_e2e.py::test_trtllm_multimodal_benchmark_serving -unittest/llmapi/apps/_test_openai_embeddings.py diff --git a/tests/integration/test_lists/qa/llm_spark_core.txt b/tests/integration/test_lists/qa/llm_spark_core.txt index e136e7706662..1234e957d2b3 100644 --- a/tests/integration/test_lists/qa/llm_spark_core.txt +++ b/tests/integration/test_lists/qa/llm_spark_core.txt @@ -17,8 +17,6 @@ test_e2e.py::test_ptp_quickstart_advanced[Qwen3-32b-nvfp4-Qwen3/nvidia-Qwen3-32B test_e2e.py::test_ptp_quickstart_advanced[Nemotron-Nano-9B-v2-nvfp4-NVIDIA-Nemotron-Nano-9B-v2-NVFP4] test_e2e.py::test_ptp_quickstart_advanced[Qwen3-30B-A3B-Qwen3/Qwen3-30B-A3B] test_e2e.py::test_ptp_quickstart_advanced[Qwen3-30B-A3B_nvfp4_hf-Qwen3/saved_models_Qwen3-30B-A3B_nvfp4_hf] -test_e2e.py::test_ptp_quickstart_advanced[Llama3.3-70B-FP8-modelopt-hf-model-hub/Llama-3.3-70B-Instruct-fp8] -test_e2e.py::test_ptp_quickstart_advanced[Llama3.3-70B-FP4-modelopt-hf-model-hub/Llama-3.3-70B-Instruct-fp4] accuracy/test_llm_api_pytorch.py::TestLlama3_1_8B::test_auto_dtype accuracy/test_llm_api_pytorch.py::TestLlama3_1_8B::test_nvfp4 diff --git a/tests/integration/test_lists/qa/llm_spark_func.yml b/tests/integration/test_lists/qa/llm_spark_func.yml index af5849e3efb6..250b7c982c5f 100644 --- a/tests/integration/test_lists/qa/llm_spark_func.yml +++ b/tests/integration/test_lists/qa/llm_spark_func.yml @@ -29,9 +29,6 @@ llm_spark_func: - test_e2e.py::test_ptp_quickstart_advanced[Phi4-Reasoning-Plus-fp8-nvidia-Phi-4-reasoning-plus-FP8] - test_e2e.py::test_ptp_quickstart_advanced[Phi4-Reasoning-Plus-nvfp4-nvidia-Phi-4-reasoning-plus-NVFP4] - test_e2e.py::test_ptp_quickstart_advanced[Phi-4-reasoning-plus-bf16-Phi-4-reasoning-plus] - - test_e2e.py::test_ptp_quickstart_advanced[Llama3.3-70B-FP8-modelopt-hf-model-hub/Llama-3.3-70B-Instruct-fp8] - - test_e2e.py::test_ptp_quickstart_advanced[Llama3.3-70B-FP4-modelopt-hf-model-hub/Llama-3.3-70B-Instruct-fp4] - - test_e2e.py::test_ptp_quickstart_advanced[Llama-4-Scout-17B-16E-FP4-llama4-models/Llama-4-Scout-17B-16E-Instruct-FP4] - test_e2e.py::test_ptp_quickstart_advanced[Nemotron-Nano-9B-v2-nvfp4-NVIDIA-Nemotron-Nano-9B-v2-NVFP4] - test_e2e.py::test_ptp_quickstart_advanced_eagle3[GPT-OSS-120B-Eagle3-gpt_oss/gpt-oss-120b-gpt_oss/gpt-oss-120b-Eagle3] - accuracy/test_llm_api_pytorch.py::TestLlama3_1_8B::test_auto_dtype @@ -64,10 +61,6 @@ llm_spark_func: gte: 2 lte: 2 tests: - - test_e2e.py::test_ptp_quickstart_advanced_multinode[llama-3.3-models/Llama-3.3-70B-Instruct-tp2] - test_e2e.py::test_ptp_quickstart_advanced_multinode[Qwen3/saved_models_Qwen3-235B-A22B_nvfp4_hf-tp2] - - test_e2e.py::test_ptp_quickstart_advanced_multinode[llama4-models/Llama-4-Scout-17B-16E-Instruct-FP8-tp2] - - test_e2e.py::test_ptp_quickstart_advanced_multinode[llama4-models/Llama-4-Scout-17B-16E-Instruct-tp2] - - accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_auto_dtype_tp2 - accuracy/test_llm_api_pytorch.py::TestQwen3_235B_A22B::test_nvfp4_2gpus[latency_moe_cutlass] - accuracy/test_llm_api_pytorch.py::TestQwen3_235B_A22B::test_nvfp4_2gpus[latency_moe_cutlass_eagle3] diff --git a/tests/integration/test_lists/qa/llm_spark_perf.yml b/tests/integration/test_lists/qa/llm_spark_perf.yml index 7dfa462ec166..2e8ae982ada9 100644 --- a/tests/integration/test_lists/qa/llm_spark_perf.yml +++ b/tests/integration/test_lists/qa/llm_spark_perf.yml @@ -34,12 +34,8 @@ llm_spark_perf: - perf/test_perf.py::test_perf[qwen3_14b_fp8-bench-pytorch-streaming-float8-maxbs:1-input_output_len:2048,128-reqs:1-con:1] - perf/test_perf.py::test_perf[qwen3_14b_fp4-bench-pytorch-streaming-float4-maxbs:1-input_output_len:2048,128-reqs:1-con:1] - perf/test_perf.py::test_perf[qwen3_14b-bench-pytorch-streaming-bfloat16-maxbs:1-input_output_len:2048,128-reqs:1-con:1] - - perf/test_perf.py::test_perf[llama_v3.3_70b_instruct_fp4-bench-pytorch-streaming-float4-maxbs:1-input_output_len:2048,128-reqs:1-con:1] - - perf/test_perf.py::test_perf[llama_v3.3_70b_instruct_fp8-bench-pytorch-streaming-float8-maxbs:1-input_output_len:2048,128-reqs:1-con:1] - perf/test_perf.py::test_perf[qwen3_30b_a3b_fp4-bench-pytorch-streaming-float4-maxbs:1-input_output_len:2048,128-reqs:1-con:1] - perf/test_perf.py::test_perf[qwen3_30b_a3b-bench-pytorch-streaming-bfloat16-maxbs:1-input_output_len:2048,128-reqs:1-con:1] - - perf/test_perf.py::test_perf[llama_v4_scout_17b_16e_instruct_fp4-bench-pytorch-streaming-float4-maxbs:1-input_output_len:2048,128-reqs:1-con:1] - - perf/test_perf.py::test_perf[llama_v3.3_nemotron_super_49b_v1.5_fp8-bench-pytorch-streaming-float8-maxbs:1-input_output_len:2048,128-reqs:1-con:1] - perf/test_perf.py::test_perf[phi_4_reasoning_plus_fp8-bench-pytorch-streaming-float8-maxbs:1-input_output_len:2048,128-reqs:1-con:1] - perf/test_perf.py::test_perf[phi_4_reasoning_plus_fp4-bench-pytorch-streaming-float4-maxbs:1-input_output_len:2048,128-reqs:1-con:1] - perf/test_perf.py::test_perf[phi_4_reasoning_plus-bench-pytorch-streaming-bfloat16-maxbs:1-input_output_len:2048,128-reqs:1-con:1] @@ -66,11 +62,8 @@ llm_spark_perf: gte: 2 lte: 2 tests: - - perf/test_perf.py::test_perf[llama_v3.3_70b_instruct-bench-pytorch-streaming-bfloat16-maxbs:1-input_output_len:2048,128-reqs:1-con:1-tp:2-gpus:2] - perf/test_perf.py::test_perf[qwen3_235b_a22b_fp4-bench-pytorch-streaming-float4-maxbs:1-input_output_len:2048,128-kv_cache_dtype:fp8-reqs:1-con:1-tp:2-gpus:2] - perf/test_perf.py::test_perf[qwen3_235b_a22b_fp4-bench-pytorch-streaming-float4-maxbs:1-input_output_len:128,2048-kv_cache_dtype:fp8-reqs:1-con:1-tp:2-gpus:2] - - perf/test_perf.py::test_perf[llama_v4_scout_17b_16e_instruct_fp8-bench-pytorch-streaming-float8-maxbs:1-input_output_len:2048,128-kv_cache_dtype:fp8-reqs:1-con:1-ep:2-tp:2-gpus:2] - - perf/test_perf.py::test_perf[llama_v4_scout_17b_16e_instruct-bench-pytorch-streaming-bfloat16-maxbs:1-input_output_len:2048,128-reqs:1-con:1-tp:2-gpus:2] # Qwen3-235B-A22B-FP4 with Eagle3 speculative decoding - perf/test_perf.py::test_perf[qwen3_235b_a22b_fp4_eagle3-bench-pytorch-streaming-float4-maxbs:1-input_output_len:2048,128-kv_cache_dtype:fp8-reqs:1-con:1-tp:2-gpus:2] - perf/test_perf.py::test_perf[qwen3_235b_a22b_fp4_eagle3-bench-pytorch-streaming-float4-maxbs:1-input_output_len:128,2048-kv_cache_dtype:fp8-reqs:1-con:1-tp:2-gpus:2] diff --git a/tests/integration/test_lists/test-db/l0_a10.yml b/tests/integration/test_lists/test-db/l0_a10.yml index e085d0c390d2..2e6dcf91b713 100644 --- a/tests/integration/test_lists/test-db/l0_a10.yml +++ b/tests/integration/test_lists/test-db/l0_a10.yml @@ -54,28 +54,6 @@ l0_a10: - unittest/tools - unittest/usage/test_transport.py - unittest/usage/test_e2e_capture.py - - disaggregated/test_disaggregated.py::test_disaggregated_single_gpu[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated.py::test_disaggregated_cuda_graph[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated.py::test_disaggregated_mixed[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated.py::test_disaggregated_overlap[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated.py::test_disaggregated_diff_max_tokens[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated.py::test_disaggregated_kv_cache_time_output[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated.py::test_disaggregated_perf_metrics[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated.py::test_disaggregated_cache_aware_balance[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated.py::test_disaggregated_conditional[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated.py::test_disaggregated_benchmark_gen_only_insufficient_kv[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated.py::test_disaggregated_ngram[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated.py::test_disaggregated_chat_completion_tool_calls[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_workers.py::test_workers_conditional_disaggregation[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_workers.py::test_workers_kv_cache_events[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_workers.py::test_workers_kv_cache_aware_router[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_workers.py::test_workers_kv_cache_aware_router_eviction[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_simple_llama[False-False-TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_simple_llama[False-True-TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_simple_llama[True-False-TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_simple_llama[True-True-TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated_single_gpu.py::test_arbitrary_kv_cache_transfer[False-TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated_single_gpu.py::test_arbitrary_kv_cache_transfer_missing_blocks[False-TinyLlama-1.1B-Chat-v1.0] - test_e2e.py::test_get_ci_container_port - test_e2e.py::test_openai_chat_multimodal_example ISOLATION - test_e2e.py::test_openai_mmencoder_example @@ -93,7 +71,25 @@ l0_a10: - test_e2e.py::test_openai_completions_example[pytorch] - test_e2e.py::test_openai_chat_example[pytorch] TIMEOUT (90) - test_e2e.py::test_trtllm_bench_request_rate_and_concurrency[enable_concurrency-] - - test_e2e.py::test_trtllm_bench_invalid_token_pytorch[TinyLlama-1.1B-Chat-v1.0-TinyLlama-1.1B-Chat-v1.0] + # ------------- Disaggregated serving tests --------------- + - disaggregated/test_disaggregated.py::test_disaggregated_single_gpu[Qwen3-0.6B] + - disaggregated/test_disaggregated.py::test_disaggregated_cuda_graph + - disaggregated/test_disaggregated.py::test_disaggregated_mixed + - disaggregated/test_disaggregated.py::test_disaggregated_overlap[Qwen3-0.6B] + - disaggregated/test_disaggregated.py::test_disaggregated_diff_max_tokens + - disaggregated/test_disaggregated.py::test_disaggregated_kv_cache_time_output + - disaggregated/test_disaggregated.py::test_disaggregated_perf_metrics + - disaggregated/test_disaggregated.py::test_disaggregated_cache_aware_balance[Qwen3-0.6B] + - disaggregated/test_disaggregated.py::test_disaggregated_conditional[Qwen3-0.6B] + - disaggregated/test_disaggregated.py::test_disaggregated_benchmark_gen_only_insufficient_kv + - disaggregated/test_disaggregated.py::test_disaggregated_ngram + - disaggregated/test_disaggregated.py::test_disaggregated_chat_completion_tool_calls[Qwen3-0.6B] + - disaggregated/test_workers.py::test_workers_conditional_disaggregation[Qwen3-0.6B] + - disaggregated/test_workers.py::test_workers_kv_cache_events[Qwen3-0.6B] + - disaggregated/test_workers.py::test_workers_kv_cache_aware_router[Qwen3-0.6B] + - disaggregated/test_workers.py::test_workers_kv_cache_aware_router_eviction[Qwen3-0.6B] + - disaggregated/test_disaggregated_single_gpu.py::test_arbitrary_kv_cache_transfer[False-Qwen3-0.6B] + - disaggregated/test_disaggregated_single_gpu.py::test_arbitrary_kv_cache_transfer_missing_blocks[False-Qwen3-0.6B] # visual_gen - unittest/_torch/visual_gen/test_profiler.py - unittest/visual_gen/test_iteration_stats.py @@ -155,8 +151,6 @@ l0_a10: stage: post_merge backend: pytorch tests: - - stress_test/stress_test.py::test_run_stress_test[llama-v3-8b-instruct-hf_tp1-stress_time_300s_timeout_450s-GUARANTEED_NO_EVICT-pytorch-stress-test] - - stress_test/stress_test.py::test_run_stress_test[llama-v3-8b-instruct-hf_tp1-stress_time_300s_timeout_450s-MAX_UTILIZATION-pytorch-stress-test] - llmapi/test_llm_examples.py::test_llmapi_chat_example - llmapi/test_llm_examples.py::test_llmapi_server_example - llmapi/test_llm_examples.py::test_llmapi_kv_cache_connector[Qwen2-0.5B] diff --git a/tests/integration/test_lists/test-db/l0_a100.yml b/tests/integration/test_lists/test-db/l0_a100.yml index 4cb0f2ac3285..bd4b57d0dd73 100644 --- a/tests/integration/test_lists/test-db/l0_a100.yml +++ b/tests/integration/test_lists/test-db/l0_a100.yml @@ -32,7 +32,6 @@ l0_a100: - accuracy/test_llm_api_pytorch_encode.py::TestEncoderEncode::test_encoder_encode_matches_huggingface_classification[bert-yelp-cuda_graph] - accuracy/test_llm_api_pytorch_encode.py::TestEncoderEncode::test_encoder_encode_cuda_graph_matches_eager_logits[bert-yelp] - accuracy/test_llm_api_pytorch_encode.py::TestEncoderEncode::test_encoder_encode_matches_huggingface_per_token_reward[qwen2.5-prm-7b] - - accuracy/test_llm_api_pytorch_encode.py::TestDecoderEncode::test_decoder_encode_matches_huggingface[tinyllama-1.1b] - accuracy/test_llm_api_pytorch_encode.py::TestDecoderEncode::test_decoder_encode_matches_huggingface[gemma-3-1b] - accuracy/test_llm_api_pytorch_encode.py::TestDecoderEncode::test_decoder_encode_matches_huggingface[phi-4-mini] - accuracy/test_llm_api_pytorch_encode.py::TestDecoderEncode::test_decoder_encode_matches_huggingface[qwen2-7b] diff --git a/tests/integration/test_lists/test-db/l0_b200.yml b/tests/integration/test_lists/test-db/l0_b200.yml index 9f9b2348103b..c7654876767e 100644 --- a/tests/integration/test_lists/test-db/l0_b200.yml +++ b/tests/integration/test_lists/test-db/l0_b200.yml @@ -65,7 +65,6 @@ l0_b200: - accuracy/test_epd_disagg_multimodal.py::TestVideoMMEEPD::test_disaggregated_videomme[qwen3vl_2b_instruct] - accuracy/test_epd_disagg_multimodal.py::TestVideoMMEEPD::test_disaggregated_videomme[nemotron_nano_v3_omni_nvfp4] - accuracy/test_llm_api_pytorch.py::TestQwen3_5_35B_A3B::test_bf16_mtp - - disaggregated/test_workers.py::test_workers_kv_cache_aware_router_eviction[TinyLlama-1.1B-Chat-v1.0] # nvbugs 5300551 - test_e2e.py::test_ptp_quickstart_advanced[Llama3.1-8B-NVFP4-nvfp4-quantized/Meta-Llama-3.1-8B] - test_e2e.py::test_ptp_quickstart_advanced[Nemotron-Nano-9B-v2-nvfp4-NVIDIA-Nemotron-Nano-9B-v2-NVFP4] - test_e2e.py::test_ptp_quickstart_advanced[Llama3.1-8B-FP8-llama-3.1-model/Llama-3.1-8B-Instruct-FP8] @@ -187,25 +186,8 @@ l0_b200: - unittest/tools/test_layer_wise_benchmarks.py::test_performance_alignment[1] - unittest/kv_cache_manager_v2_tests # ------------- KV Cache V2 Scheduler IT --------------- - - kv_cache/test_kv_cache_v2_scheduler.py::TestKVCacheV2Llama::test_v2_vs_v1_basic - - kv_cache/test_kv_cache_v2_scheduler.py::TestKVCacheV2Llama::test_token_budget_limited - - kv_cache/test_kv_cache_v2_scheduler.py::TestKVCacheV2Llama::test_chunked_prefill - - kv_cache/test_kv_cache_v2_scheduler.py::TestKVCacheV2Llama::test_chunked_prefill_multi_request - - kv_cache/test_kv_cache_v2_scheduler.py::TestKVCacheV2Llama::test_eviction[cuda_graph] - - kv_cache/test_kv_cache_v2_scheduler.py::TestKVCacheV2Llama::test_eviction[no_cuda_graph] - - kv_cache/test_kv_cache_v2_scheduler.py::TestKVCacheV2Llama::test_batch_size_limited - - kv_cache/test_kv_cache_v2_scheduler.py::TestKVCacheV2Llama::test_overlap_scheduler[non_overlap] - - kv_cache/test_kv_cache_v2_scheduler.py::TestKVCacheV2Llama::test_overlap_scheduler[overlap] - - kv_cache/test_kv_cache_v2_scheduler.py::TestKVCacheV2Llama::test_block_reuse - - kv_cache/test_kv_cache_v2_scheduler.py::TestKVCacheV2Llama::test_partial_block_reuse - - kv_cache/test_kv_cache_v2_scheduler.py::TestKVCacheV2Llama::test_chunked_prefill_with_eviction - - kv_cache/test_kv_cache_v2_scheduler.py::TestKVCacheV2Llama::test_eviction_with_block_reuse - - kv_cache/test_kv_cache_v2_scheduler.py::TestKVCacheV2Llama::test_chunked_prefill_eviction_block_reuse - - kv_cache/test_kv_cache_v2_scheduler.py::TestKVCacheV2Llama::test_eviction_overlap - - kv_cache/test_kv_cache_v2_scheduler.py::TestKVCacheV2LoRA::test_lora_v2 - - kv_cache/test_kv_cache_v2_scheduler.py::TestKVCacheV2LoRA::test_lora_multi_adapter_v2 - - kv_cache/test_kv_cache_v2_scheduler.py::TestKVCacheV2LoRA::test_lora_chunked_prefill - - kv_cache/test_kv_cache_v2_scheduler.py::TestKVCacheV2LoRA::test_lora_eviction + # ------------- KV Cache Iteration Stats --------------- + # ------------- Prefix-aware scheduling E2E tests --------------- # ------------- KV Cache Iteration Stats --------------- - kv_cache/test_kv_cache_iteration_stats.py::TestKvCacheIterationStats::test_cold_start - kv_cache/test_kv_cache_iteration_stats.py::TestKvCacheIterationStats::test_partial_block_reuse @@ -215,7 +197,6 @@ l0_b200: - kv_cache/test_kv_cache_iteration_stats.py::TestKvCacheIterationStats::test_long_context - kv_cache/test_kv_cache_iteration_stats.py::TestKvCacheIterationStats::test_rapid_fire - kv_cache/test_kv_cache_iteration_stats.py::TestKvCacheIterationStats::test_field_completeness - # ------------- Prefix-aware scheduling E2E tests --------------- - kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix_smoke # ------------- Visual Gen tests --------------- - unittest/_torch/visual_gen/test_media_decode.py @@ -441,12 +422,13 @@ l0_b200: # SM100+-gated). HW-agnostic tests (compile, models, shim, utils, plus most # custom_ops, smoke, transformations files) and pure-FP8 tests are covered # on Hopper (l0_h100.yml) and not duplicated here. + - disaggregated/test_workers.py::test_workers_kv_cache_aware_router_eviction[Qwen3-0.6B] # nvbugs 5300551 - unittest/auto_deploy/singlegpu/custom_ops/attention/test_triton_attention.py::TestSDPADispatch - unittest/auto_deploy/singlegpu/custom_ops/mamba/test_flashinfer_mamba_cached_op.py - unittest/auto_deploy/singlegpu/custom_ops/moe/test_ad_moe_op.py - unittest/auto_deploy/singlegpu/custom_ops/moe/test_trtllm_moe.py - unittest/auto_deploy/singlegpu/custom_ops/quantization/test_quant.py - - unittest/auto_deploy/singlegpu/smoke/test_ad_build_small_single.py -k "Nemotron-3-Nano-30B-A3B-FP8 or Nemotron-Nano-3-30B-A3.5B-dev or Llama-4-Scout" + - unittest/auto_deploy/singlegpu/smoke/test_ad_build_small_single.py -k "Nemotron-3-Nano-30B-A3B-FP8 or Nemotron-Nano-3-30B-A3.5B-dev" - unittest/auto_deploy/singlegpu/smoke/test_ad_speculative_decoding.py - unittest/auto_deploy/singlegpu/transformations/library/test_fuse_relu2_quant_nvfp4.py - unittest/auto_deploy/singlegpu/transformations/library/test_moe_fusion.py diff --git a/tests/integration/test_lists/test-db/l0_b200_multi_gpus_perf_sanity.yml b/tests/integration/test_lists/test-db/l0_b200_multi_gpus_perf_sanity.yml index cd6dad8198b9..06090c3d136c 100644 --- a/tests/integration/test_lists/test-db/l0_b200_multi_gpus_perf_sanity.yml +++ b/tests/integration/test_lists/test-db/l0_b200_multi_gpus_perf_sanity.yml @@ -27,6 +27,7 @@ l0_b200_multi_gpus_perf_sanity: - perf/test_perf_sanity.py::test_e2e[aggr_upload-glm5_fp4_blackwell-glm5_fp4_dep8_mtp1_8k1k] TIMEOUT (90) # gpt-oss-120b-fp4 - perf/test_perf_sanity.py::test_e2e[aggr_upload-gpt_oss_120b_fp4_blackwell-gpt_oss_fp4_tp1_mtp0_8k1k] + # qwen3.5-397b-a17b-fp4 aggregated - perf/test_perf_sanity.py::test_e2e[aggr_upload-qwen3_5_397b_fp4_blackwell-qwen3_5_397b_fp4_tp4_8k1k] - perf/test_perf_sanity.py::test_e2e[aggr_upload-qwen3_5_397b_fp4_blackwell-qwen3_5_397b_fp4_dep8_8k1k] TIMEOUT (90) - perf/test_perf_sanity.py::test_e2e[aggr_upload-qwen3_5_397b_fp4_blackwell-qwen3_5_397b_fp4_tep4_mtp3_8k1k] diff --git a/tests/integration/test_lists/test-db/l0_dgx_b200.yml b/tests/integration/test_lists/test-db/l0_dgx_b200.yml index a46a74690d77..d6c4190a1c21 100644 --- a/tests/integration/test_lists/test-db/l0_dgx_b200.yml +++ b/tests/integration/test_lists/test-db/l0_dgx_b200.yml @@ -125,10 +125,9 @@ l0_dgx_b200: - unittest/_torch/ray_orchestrator/multi_gpu/test_llm_update_weights_multi_gpu.py -m "part5" - accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_fp8_4gpus[tp4-fp8kv=False-attn_backend=TRTLLM-torch_compile=False] - accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_fp8_4gpus[tp2pp2-fp8kv=False-attn_backend=TRTLLM-torch_compile=False] - - disaggregated/test_disaggregated.py::test_disaggregated_ctxpp2_genpp2[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated.py::test_disaggregated_ctxtp2_genpp2[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated.py::test_disaggregated_ctxpp2_gentp2[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated.py::test_disaggregated_ctxpp4_gentp4[TinyLlama-1.1B-Chat-v1.0] + - disaggregated/test_disaggregated.py::test_disaggregated_ctxpp2_genpp2 + - disaggregated/test_disaggregated.py::test_disaggregated_ctxtp2_genpp2 + - disaggregated/test_disaggregated.py::test_disaggregated_ctxpp4_gentp4 - examples/test_ray.py::test_llm_inference_distributed_ray[tp2pp2] - examples/test_ray.py::test_ray_disaggregated_serving[tp2] - examples/test_ray.py::test_ray_disaggregated_serving_python[tp2] @@ -344,7 +343,6 @@ l0_dgx_b200: - accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_nvfp4_4gpus[moe_backend=CUTEDSL-mtp_nextn=2-ep4-fp8kv=True-attention_dp=True-cuda_graph=True-overlap_scheduler=True-low_precision_combine=False-torch_compile=False] - accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_nvfp4_4gpus[moe_backend=CUTEDSL-mtp_nextn=2-ep4-fp8kv=False-attention_dp=True-cuda_graph=False-overlap_scheduler=False-low_precision_combine=True-torch_compile=False] - accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_nvfp4_4gpus[moe_backend=CUTEDSL-mtp_nextn=2-ep4-fp8kv=True-attention_dp=True-cuda_graph=True-overlap_scheduler=True-low_precision_combine=True-torch_compile=False] - - accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp4_tp2pp2[torch_compile=False-enable_gemm_allreduce_fusion=False] - examples/visual_gen/test_visual_gen_wan.py::test_wan_t2v_example - examples/visual_gen/test_visual_gen_multi_gpu.py::test_wan22_t2v_lpips_against_golden_multi_gpu[ulysses4] - examples/visual_gen/test_visual_gen_multi_gpu.py::test_wan22_t2v_lpips_against_golden_multi_gpu[cfg2_ulysses2] diff --git a/tests/integration/test_lists/test-db/l0_dgx_h100.yml b/tests/integration/test_lists/test-db/l0_dgx_h100.yml index 46d80430ffe9..5c66025a6c15 100644 --- a/tests/integration/test_lists/test-db/l0_dgx_h100.yml +++ b/tests/integration/test_lists/test-db/l0_dgx_h100.yml @@ -95,7 +95,6 @@ l0_dgx_h100: - accuracy/test_llm_api_pytorch.py::TestNemotronV3Super::test_nvfp4_4gpus_hopper_w4a16 - test_e2e.py::test_ptp_quickstart_advanced_bs1 - test_e2e.py::test_ptp_quickstart_advanced_deepseek_v3_lite_4gpus_adp_balance[DeepSeek-V3-Lite-FP8-DeepSeek-V3-Lite/fp8] - - test_e2e.py::test_trtllm_bench_llmapi_launch[pytorch_backend-llama-v3-llama3-8b] # ------------- Disaggregated serving tests --------------- # Split test_py_cache_transceiver_mp.py by workflow (4 × 12 = 48 combos). # A single wrapper case ran ~3581 s on B200 and brushed the outer pytest @@ -105,17 +104,6 @@ l0_dgx_h100: - unittest/disaggregated/test_py_cache_transceiver_mp.py -k "ctx_first_sync" - unittest/disaggregated/test_py_cache_transceiver_mp.py -k "gen_first1" - unittest/disaggregated/test_py_cache_transceiver_mp.py -k "gen_first2" - - disaggregated/test_disaggregated.py::test_disaggregated_multi_gpu[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated.py::test_disaggregated_ctxpp2_genpp2[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated.py::test_disaggregated_ctxtp2_genpp2[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated.py::test_disaggregated_ctxpp2_gentp2[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated.py::test_disaggregated_ctxpp4_gentp4[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated.py::test_disaggregated_genbs1[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated.py::test_disaggregated_overlap[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated.py::test_disaggregated_overlap_transceiver_runtime_python[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated.py::test_disaggregated_python_transceiver_host_offload[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated.py::test_disaggregated_overlap_gen_first[ctx_pp1-TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated.py::test_disaggregated_overlap_gen_first[ctx_pp4-TinyLlama-1.1B-Chat-v1.0] - accuracy/test_disaggregated_serving.py::TestLlama3_1_8BInstruct::test_tp_pp_symmetric[GSM8K-tp1pp2] - accuracy/test_disaggregated_serving.py::TestLlama3_1_8BInstruct::test_tp_pp_symmetric[MMLU-tp1pp2] - accuracy/test_disaggregated_serving.py::TestLlama3_1_8BInstruct::test_tp_pp_symmetric[GSM8K-tp2pp1] @@ -218,7 +206,18 @@ l0_dgx_h100: - disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_ctxpp2_gentp2_one_mtp[DeepSeek-V3-Lite-fp8] - disaggregated/test_workers.py::test_workers_conditional_disaggregation_deepseek_v3_lite_bf16[DeepSeek-V3-Lite-bf16] - disaggregated/test_workers.py::test_workers_kv_cache_aware_router_deepseek_v3_lite_bf16[DeepSeek-V3-Lite-bf16] - - disaggregated/test_workers.py::test_workers_conversation_router[TinyLlama-1.1B-Chat-v1.0] + - disaggregated/test_disaggregated.py::test_disaggregated_ctxpp2_genpp2 + - disaggregated/test_disaggregated.py::test_disaggregated_ctxtp2_genpp2 + - disaggregated/test_disaggregated.py::test_disaggregated_ctxpp4_gentp4 + - disaggregated/test_disaggregated.py::test_disaggregated_multi_gpu[Qwen3-0.6B] + - disaggregated/test_workers.py::test_workers_conversation_router[Qwen3-0.6B] + - disaggregated/test_disaggregated.py::test_disaggregated_ctxpp4_genpp4 + - disaggregated/test_disaggregated.py::test_disaggregated_genbs1 + - disaggregated/test_disaggregated.py::test_disaggregated_overlap[Qwen3-0.6B] + - disaggregated/test_disaggregated.py::test_disaggregated_overlap_transceiver_runtime_python + - disaggregated/test_disaggregated.py::test_disaggregated_python_transceiver_host_offload + - disaggregated/test_disaggregated.py::test_disaggregated_overlap_gen_first[ctx_pp1] + - disaggregated/test_disaggregated.py::test_disaggregated_overlap_gen_first[ctx_pp4] - condition: ranges: system_gpu_count: @@ -291,11 +290,11 @@ l0_dgx_h100: - unittest/llmapi/test_llm_multi_gpu_pytorch.py -m "gpu2" - unittest/llmapi/test_async_llm.py -m "gpu2" - accuracy/test_llm_api_pytorch_ray.py::TestLlama3_1_8BInstruct::test_pp2_ray - - examples/test_ray.py::test_llm_inference_distributed_ray[tp2] - examples/test_ray.py::test_llm_inference_distributed_ray[pp2] + - examples/test_ray.py::test_llm_inference_distributed_ray[tp2pp2] - examples/test_ray.py::test_llm_inference_distributed_ray[tep2] - - examples/test_ray.py::test_ray_disaggregated_serving[tp1] - - examples/test_ray.py::test_ray_disaggregated_serving_python[tp1] + - examples/test_ray.py::test_ray_disaggregated_serving[tp2] + - examples/test_ray.py::test_ray_disaggregated_serving_python[tp2] - condition: ranges: system_gpu_count: @@ -365,13 +364,11 @@ l0_dgx_h100: auto_trigger: others orchestrator: mpi tests: - - disaggregated/test_ad_disagg_trtllm_serve.py::test_openai_completion - accuracy/test_llm_api_autodeploy.py::TestModelRegistryAccuracy::test_autodeploy_from_registry[google_gemma-3-1b-it-False] - accuracy/test_llm_api_autodeploy.py::TestModelRegistryAccuracy::test_autodeploy_from_registry[meta-llama_Llama-3.1-8B-Instruct-False] - accuracy/test_llm_api_autodeploy.py::TestModelRegistryAccuracy::test_autodeploy_from_registry[mistralai_Ministral-8B-Instruct-2410-False] - accuracy/test_llm_api_autodeploy.py::TestModelRegistryAccuracy::test_autodeploy_from_registry[mistralai_Codestral-22B-v0.1-False] - accuracy/test_llm_api_autodeploy.py::TestModelRegistryAccuracy::test_autodeploy_from_registry[Qwen_QwQ-32B-False] - - accuracy/test_llm_api_autodeploy.py::TestModelRegistryAccuracy::test_autodeploy_from_registry[meta-llama_Llama-3.3-70B-Instruct-False] - accuracy/test_llm_api_autodeploy.py::TestModelRegistryAccuracy::test_autodeploy_from_registry[nvidia_Llama-3.1-8B-Instruct-FP8-True] - accuracy/test_llm_api_autodeploy.py::TestLlama3_1_8B_Instruct_Eagle3::test_eagle3_one_model[flashinfer] - accuracy/test_llm_api_autodeploy.py::TestNemotronNanoV3::test_accuracy[bf16-4-attn_dp_off-trtllm] diff --git a/tests/integration/test_lists/test-db/l0_dgx_h200.yml b/tests/integration/test_lists/test-db/l0_dgx_h200.yml index 3b40b905220b..b870fbf61f39 100644 --- a/tests/integration/test_lists/test-db/l0_dgx_h200.yml +++ b/tests/integration/test_lists/test-db/l0_dgx_h200.yml @@ -41,12 +41,11 @@ l0_dgx_h200: - accuracy/test_disaggregated_serving.py::TestGPTOSS::test_auto_dtype[True] - accuracy/test_llm_api_pytorch.py::TestQwen3NextInstruct::test_bf16_4gpu[tep4] - accuracy/test_llm_api_pytorch.py::TestQwen3NextInstruct::test_bf16_4gpu[dep4] - - disaggregated/test_disaggregated.py::test_disaggregated_ctxtp2pp2_gentp2pp2[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated.py::test_disaggregated_ctxpp4_genpp4[TinyLlama-1.1B-Chat-v1.0] - disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_ctxtp2ep2pp2_gentp4_one_mtp_block_reuse[DeepSeek-V3-Lite-fp8] - disaggregated/test_disaggregated.py::test_disaggregated_qwen3_32b_fp8[Qwen3/Qwen3-32B-FP8] + - disaggregated/test_disaggregated.py::test_disaggregated_ctxtp2pp2_gentp2pp2 + - disaggregated/test_disaggregated.py::test_disaggregated_ctxpp4_genpp4 - disaggregated/test_disaggregated.py::test_disaggregated_mixed_stress_test[req120-conc64-qwen3_32b_fp8_mixed_stress] - - unittest/llmapi/test_llm_pytorch.py::test_nemotron_nas_lora - accuracy/test_disaggregated_serving.py::TestDeepSeekV3Lite::test_gen_only_spec_dec - condition: ranges: @@ -137,7 +136,6 @@ l0_dgx_h200: - accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_fp8_4gpus[tp2pp2-fp8kv=True-attn_backend=FLASHINFER-torch_compile=True] - accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_fp8_4gpus[pp4-fp8kv=True-attn_backend=FLASHINFER-torch_compile=False] - accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_guided_decoding_4gpus[llguidance] - - test_e2e.py::test_trtllm_bench_llmapi_launch[pytorch_backend-llama-v3-llama3-8b] - test_e2e.py::test_trtllm_bench_mgmn - accuracy/test_llm_api_pytorch.py::TestQwen3_30B_A3B_Instruct_2507::test_skip_softmax_attention_4gpus[target_sparsity_0.5-fp8kv=False] - accuracy/test_llm_api_pytorch.py::TestQwen3_30B_A3B_Instruct_2507::test_skip_softmax_attention_4gpus[target_sparsity_0.5-fp8kv=True] diff --git a/tests/integration/test_lists/test-db/l0_gb200_multi_gpus.yml b/tests/integration/test_lists/test-db/l0_gb200_multi_gpus.yml index 044926cd3f11..89496b5252d7 100644 --- a/tests/integration/test_lists/test-db/l0_gb200_multi_gpus.yml +++ b/tests/integration/test_lists/test-db/l0_gb200_multi_gpus.yml @@ -34,11 +34,6 @@ l0_gb200_multi_gpus: - accuracy/test_llm_api_pytorch.py::TestNemotronV3Ultra::test_nvfp4_4gpus_block_reuse[ADP4_MTP] - accuracy/test_llm_api_pytorch.py::TestQwen3_5_397B_A17B::test_nvfp4_4gpus_online_eplb[moe_backend=CUTEDSL] - accuracy/test_llm_api_pytorch.py::TestQwen3_5_397B_A17B::test_nvfp4_4gpus_online_eplb[moe_backend=TRTLLM] - - accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp8_tp4[torch_compile=False] - - accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp8_tp4[torch_compile=True] - - accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_nvfp4_tp4[torch_compile=False] - - accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_nvfp4_tp4[torch_compile=True] - - accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp4_tp2pp2[torch_compile=True-enable_gemm_allreduce_fusion=False] - accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_w4_4gpus[v2_kv_cache-tp4-trtllm-auto] - accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_w4_4gpus[v2_kv_cache-dp4-cutlass-auto] - accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_w4_4gpus_online_eplb[fp8] @@ -55,8 +50,8 @@ l0_gb200_multi_gpus: - unittest/_torch/modules/moe/test_moe_comm.py::TestMoEComm::test_moe_comm - unittest/_torch/modules/moe/test_moe_comm.py::TestMoEComm::test_nccl_ep_cuda_graph_replay_uses_updated_routing - unittest/_torch/modules/moe/test_moe_comm.py::TestMoEComm::test_moe_comm_postquant - - disaggregated/test_disaggregated.py::test_disaggregated_overlap_transceiver_runtime_python_fabric_memory[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated.py::test_disaggregated_overlap_transceiver_runtime_python_bounce[TinyLlama-1.1B-Chat-v1.0] + - disaggregated/test_disaggregated.py::test_disaggregated_overlap_transceiver_runtime_python_fabric_memory + - disaggregated/test_disaggregated.py::test_disaggregated_overlap_transceiver_runtime_python_bounce - condition: ranges: diff --git a/tests/integration/test_lists/test-db/l0_h100.yml b/tests/integration/test_lists/test-db/l0_h100.yml index 8023c33f927c..118e69cc8a7a 100644 --- a/tests/integration/test_lists/test-db/l0_h100.yml +++ b/tests/integration/test_lists/test-db/l0_h100.yml @@ -180,7 +180,6 @@ l0_h100: - disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_tp1_single_gpu_mtp[DeepSeek-V3-Lite-fp8] - disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_tp1_two_mtp[DeepSeek-V3-Lite-fp8] - disaggregated/test_disaggregated.py::test_disaggregated_load_balance[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated.py::test_disaggregated_tinyllama_multi_orchestrator[TinyLlama-1.1B-Chat-v1.0] - disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_simple_deepseek[False-False-DeepSeek-V3-Lite-fp8/fp8] - disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_simple_deepseek[False-True-DeepSeek-V3-Lite-fp8/fp8] - disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_simple_deepseek[True-False-DeepSeek-V3-Lite-fp8/fp8] @@ -192,11 +191,6 @@ l0_h100: - disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_llama_context_capacity[False-False-DeepSeek-V3-Lite-fp8/fp8] - disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_spec_dec_batch_slot_limit[True-False-EAGLE3-LLaMA3.1-Instruct-8B-Llama-3.1-8B-Instruct] - disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_spec_dec_batch_slot_limit[False-False-EAGLE3-LLaMA3.1-Instruct-8B-Llama-3.1-8B-Instruct] - - disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_cancel_gen_requests[TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_logits[False-TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_logits[True-TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_logprobs[False-TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_logprobs[True-TinyLlama-1.1B-Chat-v1.0] # Encoder-decoder Hopper smoke: CUDA-graph beam/greedy, kv-v2, overlap. # The primary pre-merge set runs on L40S (l0_l40s.yml); the full # dtype/model-size matrix runs post-merge below. @@ -253,13 +247,8 @@ l0_h100: - accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_fp8_block_scales[mtp=disable-fp8kv=True-attention_dp=False-cuda_graph=True-overlap_scheduler=True-torch_compile=True] - accuracy/test_llm_api_pytorch.py::TestQwen3_8B::test_fp8_block_scales[latency] - test_e2e.py::test_trtllm_bench_pytorch_backend_sanity[meta-llama/Llama-3.1-8B-llama-3.1-8b-False-False] - - disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_simple_llama[False-False-TinyLlama-1.1B-Chat-v1.0] - disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_simple_deepseek[False-False-DeepSeek-V3-Lite-fp8/fp8] - disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_simple_qwen3[False-False-Qwen3-8B-FP8] - - disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_logits[False-TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_logits[True-TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_logprobs[False-TinyLlama-1.1B-Chat-v1.0] - - disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_logprobs[True-TinyLlama-1.1B-Chat-v1.0] - unittest/_torch/executor/test_overlap_scheduler.py - unittest/executor/test_shim_ray.py - unittest/_torch/ray_orchestrator/single_gpu/test_llm_sleep.py @@ -503,12 +492,13 @@ l0_h100: - examples/test_ad_speculative_decoding.py::test_eagle_wrapper_forward[2] - examples/test_ad_speculative_decoding.py::test_nemotron_mtp_model_with_weights - examples/test_ad_guided_decoding.py::test_autodeploy_guided_decoding_main_json - - disaggregated/test_ad_disagg.py::test_disaggregated_logits[tinyllama] + - disaggregated/test_disaggregated.py::test_disaggregated_load_balance + - disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_cancel_gen_requests[Qwen3-0.6B] + - disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_logprobs[False-Qwen3-0.6B] + - disaggregated/test_disaggregated_single_gpu.py::test_disaggregated_logprobs[True-Qwen3-0.6B] + - disaggregated/test_ad_disagg.py::test_disaggregated_logits[qwen3_0.6b] - disaggregated/test_ad_disagg.py::test_disaggregated_logits[deepseek_v3_mla] - - disaggregated/test_ad_disagg.py::test_reduced_layer_handoff_matches_aggregate[tinyllama] - disaggregated/test_ad_disagg.py::test_reduced_layer_handoff_matches_aggregate[deepseek_v3_mla] - - disaggregated/test_ad_disagg.py::test_tinyllama_batch_handoff_semantic_slots - - disaggregated/test_ad_disagg.py::test_chunked_prefill_handoff[tinyllama] - disaggregated/test_ad_disagg.py::test_chunked_prefill_handoff[deepseek_v3_mla] - condition: ranges: diff --git a/tests/integration/test_lists/test-db/l0_l40s.yml b/tests/integration/test_lists/test-db/l0_l40s.yml index 4deb7d3893c3..095f12922833 100644 --- a/tests/integration/test_lists/test-db/l0_l40s.yml +++ b/tests/integration/test_lists/test-db/l0_l40s.yml @@ -25,7 +25,6 @@ l0_l40s: - unittest/_torch/modeling/test_modeling_step3p7vl.py - unittest/llmapi/apps/_test_openai_chat_multimodal.py::test_single_chat_session_image_embeds -m needs_l40s # Encoder-only /v1/embeddings dynamic batching (BERT classifier + PRM-7B reward + Qwen3-Embedding) - - unittest/llmapi/apps/_test_openai_embeddings.py # Qwen3-Embedding text-embedding accuracy vs HuggingFace (0.6B small/fast + 8B large) - accuracy/test_llm_api_pytorch_encode.py::TestEncoderEncode::test_qwen3_text_embedding_matches_huggingface[qwen3-embedding-0.6b] - accuracy/test_llm_api_pytorch_encode.py::TestEncoderEncode::test_qwen3_text_embedding_matches_huggingface[qwen3-embedding-8b] @@ -96,7 +95,7 @@ l0_l40s: - llmapi/test_llm_examples.py::test_llmapi_example_multilora - llmapi/test_llm_examples.py::test_llmapi_example_guided_decoding - llmapi/test_llm_examples.py::test_llmapi_example_logits_processor - - examples/test_llm_api_with_mpi.py::test_llm_api_single_gpu_with_mpirun[TinyLlama-1.1B-Chat-v1.0] + - examples/test_llm_api_with_mpi.py::test_llm_api_single_gpu_with_mpirun[Qwen3-0.6B] - condition: ranges: system_gpu_count: diff --git a/tests/integration/test_lists/test-db/l0_sanity_check.yml b/tests/integration/test_lists/test-db/l0_sanity_check.yml index db5c37afffbe..ecf2de715787 100644 --- a/tests/integration/test_lists/test-db/l0_sanity_check.yml +++ b/tests/integration/test_lists/test-db/l0_sanity_check.yml @@ -28,7 +28,6 @@ l0_sanity_check: - llmapi/test_llm_examples.py::test_llmapi_example_logits_processor - llmapi/test_llm_examples.py::test_llmapi_sampling - llmapi/test_llm_examples.py::test_llmapi_runtime - - examples/test_llm_api_with_mpi.py::test_llm_api_single_gpu_with_mpirun[TinyLlama-1.1B-Chat-v1.0] ISOLATION - unittest/others/test_kv_cache_transceiver.py::test_kv_cache_transceiver_single_process[NIXL-mha-ctx_fp16_gen_fp16] - unittest/others/test_kv_cache_transceiver.py::test_kv_cache_transceiver_single_process[UCX-mha-ctx_fp16_gen_fp16] - unittest/others/test_kv_cache_transceiver.py::test_cpp_nixl_sync_transfer_stress @@ -39,6 +38,7 @@ l0_sanity_check: - unittest/others/test_kv_cache_transceiver.py::test_kv_transfer_timeout_silent_when_unset - unittest/_torch/executor/test_model_loader_mx.py - unittest/_torch/executor/test_hang_detector_kill.py + - examples/test_llm_api_with_mpi.py::test_llm_api_single_gpu_with_mpirun[Qwen3-0.6B] ISOLATION - condition: ranges: system_gpu_count: diff --git a/tests/integration/test_lists/waives.txt b/tests/integration/test_lists/waives.txt index a11a8e3b1502..6e7379be69fd 100644 --- a/tests/integration/test_lists/waives.txt +++ b/tests/integration/test_lists/waives.txt @@ -61,9 +61,9 @@ accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_bfloat16_4gpus[t accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_bfloat16_4gpus[tp4-attn_backend=TRTLLM-torch_compile=False] SKIP (https://nvbugs/5616182) accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_fp8_4gpus[pp4-fp8kv=False-attn_backend=TRTLLM-torch_compile=False] SKIP (https://nvbugs/6437412) accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_fp8_4gpus[pp4-fp8kv=True-attn_backend=TRTLLM-torch_compile=False] SKIP (https://nvbugs/6278337) -accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp4_tp2pp2[torch_compile=False-enable_gemm_allreduce_fusion=False] SKIP (https://nvbugs/6428089) -accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_fp4_tp2pp2[torch_compile=True-enable_gemm_allreduce_fusion=True] SKIP (https://nvbugs/6211441) -accuracy/test_llm_api_pytorch.py::TestNemotronV3Super::test_nvfp4_4gpus_hopper_w4a16 SKIP (https://nvbugs/6478723) +accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_fp8_4gpus[tp2pp2-fp8kv=False-attn_backend=TRTLLM-torch_compile=False] SKIP (https://nvbugs/6427411) +accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_fp8_4gpus[tp2pp2-fp8kv=True-attn_backend=TRTLLM-torch_compile=False] SKIP (https://nvbugs/6427411) +accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_guided_decoding_4gpus[xgrammar] SKIP (https://nvbugs/6427411) accuracy/test_llm_api_pytorch.py::TestNemotronV3Super::test_nvfp4_8gpus_mtp SKIP (https://nvbugs/6581065) accuracy/test_llm_api_pytorch.py::TestNemotronV3Ultra::test_nvfp4_4gpus_static_eplb[moe_backend=CUTLASS] SKIP (https://nvbugs/6535767) accuracy/test_llm_api_pytorch.py::TestQwen3NextInstruct::test_nvfp4[tp1_block_reuse-cutlass] SKIP (https://nvbugs/6535767) @@ -85,8 +85,6 @@ cpp/test_multi_gpu.py::test_cache_transceiver[8proc-mooncake_kvcache-90] SKIP (h cpp/test_multi_gpu.py::test_cache_transceiver[8proc-ucx_kvcache-90] SKIP (https://nvbugs/5838199) disaggregated/test_auto_scaling.py::test_disagg_server_restart[etcd-round_robin] SKIP (https://nvbugs/6611817) disaggregated/test_disaggregated.py::test_disaggregated_cancel_large_context_requests[DeepSeek-V3-Lite-bf16] SKIP (https://nvbugs/6105768) -disaggregated/test_disaggregated.py::test_disaggregated_ctxpp4_genpp4[TinyLlama-1.1B-Chat-v1.0] SKIP (https://nvbugs/6428069) -disaggregated/test_disaggregated.py::test_disaggregated_ctxtp2pp2_gentp2pp2[TinyLlama-1.1B-Chat-v1.0] SKIP (https://nvbugs/6428069) disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_bf16_cache_aware_balance[DeepSeek-V3-Lite-bf16] SKIP (https://nvbugs/6162322) disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_bf16_conditional[DeepSeek-V3-Lite-bf16] SKIP (https://nvbugs/6162322) disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_attention_dp_gen_only[DeepSeek-V3-Lite-fp8] SKIP (https://nvbugs/6162322) @@ -101,14 +99,12 @@ disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_tp1 disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_tp1_single_gpu[DeepSeek-V3-Lite-fp8] SKIP (https://nvbugs/6162322) disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_tp1_single_gpu_mtp[DeepSeek-V3-Lite-fp8] SKIP (https://nvbugs/6162322) disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_tp1_two_mtp[DeepSeek-V3-Lite-fp8] SKIP (https://nvbugs/6162322) -disaggregated/test_disaggregated.py::test_disaggregated_genbs1[TinyLlama-1.1B-Chat-v1.0] SKIP (https://nvbugs/6162322) disaggregated/test_disaggregated.py::test_disaggregated_qwen3_32b_fp8[Qwen3/Qwen3-32B-FP8] SKIP (https://nvbugs/6566734) disaggregated/test_disaggregated.py::test_disaggregated_stress_test[input8k-output1k-conc512-deepseek_r1_v2_fp4_stress] SKIP (https://nvbugs/6621358) disaggregated/test_disaggregated.py::test_disaggregated_stress_test[input8k-output1k-conc512-gpt_oss_120b_eagle_triton_stress] SKIP (https://nvbugs/6621362) disaggregated/test_disaggregated.py::test_disaggregated_stress_test[input8k-output1k-conc512-qwen3_5_4b_fp8_stress] SKIP (https://nvbugs/6621362) disaggregated/test_workers.py::test_workers_conversation_router[TinyLlama-1.1B-Chat-v1.0] SKIP (https://nvbugs/6162322) disaggregated/test_workers.py::test_workers_kv_cache_aware_router_deepseek_v3_lite_bf16[DeepSeek-V3-Lite-bf16] SKIP (https://nvbugs/6162322) -disaggregated/test_workers.py::test_workers_kv_cache_aware_router_eviction[TinyLlama-1.1B-Chat-v1.0] SKIP (https://nvbugs/6162322) examples/test_ad_speculative_decoding.py::test_autodeploy_eagle3_one_model_acceptance_rate[trtllm-torch-cudagraph] SKIP (https://nvbugs/6426841) examples/test_ray.py::test_ray_disaggregated_serving[tp2] SKIP (https://nvbugs/6601575) examples/test_ray.py::test_ray_disaggregated_serving_python[tp2] SKIP (https://nvbugs/6601574) @@ -138,7 +134,6 @@ full:B200/accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_cute_dsl_nv full:B200/accuracy/test_llm_api_pytorch.py::TestLagunaXS::test_fp8 SKIP (https://nvbugs/6525011) full:B200/accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_fp8[fp8kv=True-attn_backend=TRTLLM-torch_compile=True] SKIP (https://nvbugs/6473161) full:B200/accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_fp8_4gpus[tp4-fp8kv=True-attn_backend=TRTLLM-torch_compile=True] SKIP (https://nvbugs/6473161) -full:B200/accuracy/test_llm_api_pytorch.py::TestLlama3_3_70BInstruct::test_nvfp4_tp4[torch_compile=True] SKIP (https://nvbugs/6525010) full:B200/accuracy/test_llm_api_pytorch.py::TestMiniMaxM3::test_auto_dtype[tp_size=8-ep_size=8] SKIP (https://nvbugs/6384747) full:B200/accuracy/test_llm_api_pytorch.py::TestMiniMaxM3::test_mxfp8[use_msa=False] SKIP (https://nvbugs/6424188) full:B200/accuracy/test_llm_api_pytorch.py::TestMiniMaxM3::test_nvfp4[use_msa=False] SKIP (https://nvbugs/6424188) @@ -171,10 +166,8 @@ full:DGX_B200/accuracy/test_llm_api_pytorch.py::TestDeepSeekV4Pro::test_gsm8k_fu full:DGX_B200/disaggregated/test_disaggregated.py::test_disaggregated_gpt_oss_120b_harmony[gpt_oss/gpt-oss-120b] SKIP (https://nvbugs/6594241) full:DGX_B200/perf/test_perf_sanity.py::test_e2e[aggr_upload-gemma4_26b_a4b_nvfp4_blackwell-gemma4_26b_a4b_nvfp4_tp1_1k1k] SKIP (https://nvbugs/6571410) full:DGX_B200/perf/test_perf_sanity.py::test_e2e[aggr_upload-host_perf_llama8b_spec_decode-llama8b_spec_bs1_128_128] SKIP (https://nvbugs/6571408) -full:DGX_B200/unittest/llmapi/test_llm_multi_gpu_pytorch.py::test_tinyllama_logits_processor_tp2pp2 SKIP (https://nvbugs/6618096) full:DGX_H100/unittest/llmapi/test_llm_multi_gpu_pytorch.py -m "gpu4" SKIP (https://nvbugs/6618102) full:DGX_H100/unittest/llmapi/test_llm_multi_gpu_pytorch.py::test_llm_get_stats_pp4[False-False-True] SKIP (https://nvbugs/6618098) -full:DGX_H100/unittest/llmapi/test_llm_multi_gpu_pytorch.py::test_tinyllama_logits_processor_tp2pp2 SKIP (https://nvbugs/6618106) full:GB200/accuracy/test_disaggregated_serving.py::TestLlama3_1_8BInstruct::test_auto_dtype[ctx_block_reuse_only] SKIP (https://nvbugs/6525893) full:GB200/accuracy/test_dwdp_disaggregated_serving.py::TestDwdpDeepSeekV3Lite::test_dwdp_accuracy SKIP (https://nvbugs/6276923) full:GB200/accuracy/test_dwdp_disaggregated_serving.py::TestDwdpDeepSeekV3Lite::test_dwdp_accuracy_contention_opt SKIP (https://nvbugs/6276923) @@ -194,7 +187,6 @@ full:GB200/disaggregated/test_ad_disagg.py::test_async_eagle3_full_model_handoff full:GB200/disaggregated/test_ad_disagg.py::test_async_generation_matches_aggregate SKIP (https://nvbugs/6415323) full:GB200/disaggregated/test_ad_disagg.py::test_async_generation_no_overlap_matches_aggregate SKIP (https://nvbugs/6402495) full:GB200/disaggregated/test_ad_disagg.py::test_async_sharded_generation_handoff SKIP (https://nvbugs/6402495) -full:GB200/disaggregated/test_ad_disagg_trtllm_serve.py::test_openai_completion SKIP (https://nvbugs/6402495) full:GB200/llmapi/test_llm_api_pytorch_moe_lora.py::test_qwen_moe_routed_expert_multi_lora_varying_ranks[cudagraph] SKIP (https://nvbugs/6475623) full:GB300/accuracy/test_disaggregated_serving.py::TestDeepSeekV3Lite::test_gen_first[adp-mtp2] SKIP (https://nvbugs/6295740) full:GB300/accuracy/test_llm_api_autodeploy.py::TestNemotronNanoV3::test_accuracy[bf16-4-attn_dp_off-trtllm] SKIP (https://nvbugs/6539942) @@ -223,7 +215,6 @@ full:H100/accuracy/test_llm_api_pytorch.py::TestQwen3_30B_A3B_Instruct_2507::tes full:H100/accuracy/test_llm_api_pytorch.py::TestQwen3_5_35B_A3B::test_bf16[tp1-CUTLASS] SKIP (https://nvbugs/6273850) full:H100/disaggregated/test_disaggregated.py::test_disaggregated_logprobs_serving[llama-3.1-8b-instruct] SKIP (https://nvbugs/6275959) full:H100/disaggregated/test_disaggregated.py::test_disaggregated_stress_test[input8k-output1k-conc512-qwen3_32b_fp8_stress] SKIP (https://nvbugs/6312828) -full:H100_PCIe/unittest/llmapi/test_llm_pytorch.py::test_llama_7b_multi_lora_evict_and_reload_lora_gpu_cache SKIP (https://nvbugs/5682551) full:H20/accuracy/test_disaggregated_serving.py::TestDeepSeekV3Lite::test_auto_dtype[mtp_nextn=2-overlap_scheduler=False] SKIP (https://nvbugs/6345827) full:H20/accuracy/test_disaggregated_serving.py::TestDeepSeekV3Lite::test_auto_dtype[mtp_nextn=2-overlap_scheduler=True] SKIP (https://nvbugs/6345827) full:H20/accuracy/test_epd_disagg_multimodal.py::TestVideoMMEEPD::test_disaggregated_videomme[nemotron_nano_v3_omni_fp8] SKIP (https://nvbugs/6327718) @@ -271,13 +262,6 @@ full:RTX_PRO_6000_Blackwell_Server_Edition/accuracy/test_llm_api_pytorch.py::Tes full:RTX_PRO_6000_Blackwell_Server_Edition/accuracy/test_llm_api_pytorch.py::TestQwen3_5_4B::test_dflash SKIP (https://nvbugs/6273850) full:RTX_PRO_6000_Blackwell_Server_Edition/accuracy/test_llm_api_pytorch.py::TestQwen3_5_4B::test_fp8 SKIP (https://nvbugs/6273850) full:sm100/unittest/bindings SKIP (Disable for Blackwell) -kv_cache/test_kv_cache_v2_scheduler.py::TestKVCacheV2Llama::test_chunked_prefill SKIP (https://nvbugs/6428002) -kv_cache/test_kv_cache_v2_scheduler.py::TestKVCacheV2Llama::test_chunked_prefill_eviction_block_reuse SKIP (https://nvbugs/6607481) -kv_cache/test_kv_cache_v2_scheduler.py::TestKVCacheV2Llama::test_eviction[cuda_graph] SKIP (https://nvbugs/6600098) -kv_cache/test_kv_cache_v2_scheduler.py::TestKVCacheV2Llama::test_eviction_with_block_reuse SKIP (https://nvbugs/6462303) -kv_cache/test_kv_cache_v2_scheduler.py::TestKVCacheV2Llama::test_overlap_scheduler[non_overlap] SKIP (https://nvbugs/6600098) -kv_cache/test_kv_cache_v2_scheduler.py::TestKVCacheV2Llama::test_overlap_scheduler[overlap] SKIP (https://nvbugs/6600098) -kv_cache/test_kv_cache_v2_scheduler.py::TestKVCacheV2Llama::test_token_budget_limited SKIP (https://nvbugs/6600098) llmapi/test_llm_api_pytorch_bart.py::test_bart_pytorch_generate_encoder_decoder_end_to_end[bf16-kv-v1-cuda-graph-on-greedy-tp2-bart-large-cnn] SKIP (https://nvbugs/6463812) llmapi/test_llm_api_pytorch_moe_lora.py::test_mixtral_moe_routed_expert_fp8_multi_lora_varying_ranks[cudagraph] SKIP (https://nvbugs/6463829) llmapi/test_llm_api_pytorch_moe_lora.py::test_mixtral_moe_routed_expert_fp8_multi_lora_varying_ranks[eager] SKIP (https://nvbugs/6463829) @@ -311,7 +295,6 @@ test_e2e.py::test_multi_nodes_eval[MiniMax-M3-tp16-mmlu] SKIP (https://nvbugs/63 test_e2e.py::test_ptp_quickstart_advanced_deepseek_r1_w4afp8_8gpus[DeepSeek-R1-W4AFP8-DeepSeek-R1/DeepSeek-R1-W4AFP8] SKIP (https://nvbugs/5836830) test_e2e.py::test_ptp_quickstart_bert[TRTLLM-BertForSequenceClassification-bert/bert-base-uncased-yelp-polarity] SKIP (https://nvbugs/6605819) test_e2e.py::test_ptp_quickstart_bert[VANILLA-BertForSequenceClassification-bert/bert-base-uncased-yelp-polarity] SKIP (bug pending, tracked in PR 17414) -test_e2e.py::test_trtllm_bench_llmapi_launch[pytorch_backend-llama-v3-llama3-8b] SKIP (https://nvbugs/6568058) unittest/_torch/attention/sparse/dsa/test_req_idx_per_token.py::test_on_update_kv_lens_rebuilds_stale_map SKIP (https://nvbugs/6574939) unittest/_torch/attention/sparse/rocketkv/test_rocketkv.py::test_model[TRTLLM-llama-3.1-model/Llama-3.1-8B-Instruct-pytorch] SKIP (https://nvbugs/6602094) unittest/_torch/attention/sparse/rocketkv/test_rocketkv.py::test_model[VANILLA-llama-3.1-model/Llama-3.1-8B-Instruct-pytorch] SKIP (https://nvbugs/6602094) diff --git a/tests/scripts/perf-sanity/aggregated/llama_v3_3_70b_instruct_fp4_blackwell.yaml b/tests/scripts/perf-sanity/aggregated/llama_v3_3_70b_instruct_fp4_blackwell.yaml index ce97ab837d5e..351c338310c6 100644 --- a/tests/scripts/perf-sanity/aggregated/llama_v3_3_70b_instruct_fp4_blackwell.yaml +++ b/tests/scripts/perf-sanity/aggregated/llama_v3_3_70b_instruct_fp4_blackwell.yaml @@ -1,5 +1,4 @@ metadata: - model_name: llama_v3.3_70b_instruct_fp4 supported_gpus: - B200 hardware: @@ -7,7 +6,6 @@ hardware: server_configs: # TP4, ISL/OSL: 512/32 - name: "llama70b_fp4_tp4_512_32" - model_name: "llama_v3.3_70b_instruct_fp4" tensor_parallel_size: 4 pipeline_parallel_size: 1 max_batch_size: 512 @@ -28,7 +26,6 @@ server_configs: # TP4, ISL/OSL: 1000/1000 - name: "llama70b_fp4_tp4_1000_1000" - model_name: "llama_v3.3_70b_instruct_fp4" tensor_parallel_size: 4 pipeline_parallel_size: 1 max_batch_size: 1024 diff --git a/tests/test_common/llm_data.py b/tests/test_common/llm_data.py index 6605613df19e..2ca9272f6ce9 100644 --- a/tests/test_common/llm_data.py +++ b/tests/test_common/llm_data.py @@ -31,8 +31,6 @@ "meta-llama/Llama-3.1-8B": "llama-3.1-model/Meta-Llama-3.1-8B", "nvidia/Llama-3.1-8B-Instruct-FP8": "Llama-3.1-8B-Instruct-FP8", "nvidia/Llama-3.1-8B-Instruct-NVFP4": "Llama-3.1-8B-Instruct-NVFP4", - "TinyLlama/TinyLlama-1.1B-Chat-v1.0": "llama-models-v2/TinyLlama-1.1B-Chat-v1.0", - "meta-llama/Llama-4-Scout-17B-16E-Instruct": "llama4-models/Llama-4-Scout-17B-16E-Instruct", "mistralai/Mixtral-8x7B-Instruct-v0.1": "Mixtral-8x7B-Instruct-v0.1", "mistralai/Mistral-Small-3.1-24B-Instruct-2503": "Mistral-Small-3.1-24B-Instruct-2503", "Qwen/Qwen3-30B-A3B": "Qwen3/Qwen3-30B-A3B", @@ -62,10 +60,8 @@ "google/gemma-4-E2B-it": "gemma/gemma-4-E2B-it", "nvidia/Qwen3.5-397B-A17B-NVFP4": "Qwen3.5-397B-A17B-NVFP4", "Qwen/QwQ-32B": "QwQ-32B", - "meta-llama/Llama-3.3-70B-Instruct": "llama-3.3-models/Llama-3.3-70B-Instruct", "mistralai/Codestral-22B-v0.1": "Codestral-22B-v0.1", "mistralai/Ministral-8B-Instruct-2410": "Ministral-8B-Instruct-2410", - "nvidia/Llama-3.1-Nemotron-Nano-8B-v1": "Llama-3.1-Nemotron-Nano-8B-v1", "google/gemma-4-26B-A4B-it": "gemma/gemma-4-26B-A4B-it", "Qwen/Qwen3.5-35B-A3B": "Qwen3.5-35B-A3B", "nvidia/Cosmos3-Nano": "nvidia/Cosmos3-Nano", diff --git a/tests/unittest/_torch/multi_gpu/test_mpi_sleep_wakeup.py b/tests/unittest/_torch/multi_gpu/test_mpi_sleep_wakeup.py index 3484aa0cdedc..2799bc301a9d 100644 --- a/tests/unittest/_torch/multi_gpu/test_mpi_sleep_wakeup.py +++ b/tests/unittest/_torch/multi_gpu/test_mpi_sleep_wakeup.py @@ -15,7 +15,7 @@ """Multi-rank (TP=2) sleep/wakeup tests for the MPI/IPC executor path. Verifies that sleep() and wakeup() correctly release and restore GPU memory on -*all* ranks, not just rank-0. Uses TinyLlama with tensor_parallel_size=2 so +*all* ranks, not just rank-0. Uses Llama-3.1-8B with tensor_parallel_size=2 so the PyExecutor starts two MPI worker processes; the control-listener thread on rank-1 is exercised by every sleep/wakeup call. @@ -33,7 +33,7 @@ from tensorrt_llm.llmapi import KvCacheConfig, SamplingParams from tensorrt_llm.llmapi.llm_args import ExecutorMemoryType, SleepConfig -_LLAMA_MODEL_PATH = str(llm_models_root() / "llama-models-v2/TinyLlama-1.1B-Chat-v1.0") +_LLAMA_MODEL_PATH = str(llm_models_root() / "llama-3.1-model/Llama-3.1-8B-Instruct") _PROMPTS = [ "Hello, my name is", diff --git a/tests/unittest/_torch/ray_orchestrator/multi_gpu/test_inflight_weight_update.py b/tests/unittest/_torch/ray_orchestrator/multi_gpu/test_inflight_weight_update.py index 0577588aaeaa..c885fb5f9c08 100644 --- a/tests/unittest/_torch/ray_orchestrator/multi_gpu/test_inflight_weight_update.py +++ b/tests/unittest/_torch/ray_orchestrator/multi_gpu/test_inflight_weight_update.py @@ -98,7 +98,7 @@ async def _run_generate_async( @pytest.mark.asyncio @skip_pre_hopper async def test_inflight_weight_update(): - model_dir = str(llm_models_root() / "llama-models-v2/TinyLlama-1.1B-Chat-v1.0") + model_dir = str(llm_models_root() / "llama-3.1-model/Llama-3.1-8B-Instruct") num_hidden_layers = 1 # Reference HF model providing the "new" weights via CUDA IPC handles. diff --git a/tests/unittest/_torch/ray_orchestrator/single_gpu/test_llm_sleep.py b/tests/unittest/_torch/ray_orchestrator/single_gpu/test_llm_sleep.py index 9a812bbb95c9..9f8436c3bc3c 100644 --- a/tests/unittest/_torch/ray_orchestrator/single_gpu/test_llm_sleep.py +++ b/tests/unittest/_torch/ray_orchestrator/single_gpu/test_llm_sleep.py @@ -7,7 +7,7 @@ def test_llm_sleep(process_gpu_memory_info_available): - llama_model_path = str(llm_models_root() / "llama-models-v2/TinyLlama-1.1B-Chat-v1.0") + llama_model_path = str(llm_models_root() / "llama-3.1-model/Llama-3.1-8B-Instruct") kv_cache_config = KvCacheConfig(enable_block_reuse=False, max_tokens=16384) llm = LLM( @@ -73,7 +73,7 @@ def test_llm_sleep_discard_weights(process_gpu_memory_info_available): are gone (NONE = no backup). The model should still be able to run a forward pass without crashing — output correctness is not expected. """ - llama_model_path = str(llm_models_root() / "llama-models-v2/TinyLlama-1.1B-Chat-v1.0") + llama_model_path = str(llm_models_root() / "llama-3.1-model/Llama-3.1-8B-Instruct") kv_cache_config = KvCacheConfig(enable_block_reuse=False, max_tokens=16384) sleep_config = SleepConfig( diff --git a/tests/unittest/_torch/sampler/test_beam_search.py b/tests/unittest/_torch/sampler/test_beam_search.py index b615a5bf3b5c..1bba69f8889e 100644 --- a/tests/unittest/_torch/sampler/test_beam_search.py +++ b/tests/unittest/_torch/sampler/test_beam_search.py @@ -2287,8 +2287,7 @@ def sampler_type(request) -> str: def model_kwargs() -> dict[str, Any]: root = llm_models_root() assert root is not None - return dict(model=root / "llama-models-v2" / - "TinyLlama-1.1B-Chat-v1.0", ) + return dict(model=root / "llama-3.1-model" / "Llama-3.1-8B-Instruct", ) # NB: Class-level fixture overrides do not work without this @pytest.fixture(scope="module") diff --git a/tests/unittest/_torch/sampler/test_best_of_n.py b/tests/unittest/_torch/sampler/test_best_of_n.py index 90890efdd904..3b3d3ca13ffb 100644 --- a/tests/unittest/_torch/sampler/test_best_of_n.py +++ b/tests/unittest/_torch/sampler/test_best_of_n.py @@ -34,8 +34,8 @@ def expected_outputs(): @pytest.fixture(scope="module") def llm(): - return LLM(model=os.path.join(llm_models_root(), "llama-models-v2", - "TinyLlama-1.1B-Chat-v1.0"), + return LLM(model=os.path.join(llm_models_root(), "llama-3.1-model", + "Llama-3.1-8B-Instruct"), kv_cache_config=KvCacheConfig(max_tokens=1000), max_batch_size=8, max_seq_len=64, diff --git a/tests/unittest/_torch/sampler/test_logits_logprobs.py b/tests/unittest/_torch/sampler/test_logits_logprobs.py index 9637d8b47c8c..a4ca09eafa5a 100644 --- a/tests/unittest/_torch/sampler/test_logits_logprobs.py +++ b/tests/unittest/_torch/sampler/test_logits_logprobs.py @@ -106,7 +106,7 @@ def llm( ) llm = LLM( - model=os.path.join(llm_models_root(), "llama-models-v2", "TinyLlama-1.1B-Chat-v1.0"), + model=os.path.join(llm_models_root(), "llama-3.1-model", "Llama-3.1-8B-Instruct"), kv_cache_config=global_kvcache_config, max_batch_size=128, # reduce buffer sizes, specially for generation logits sampler_type=sampler_type, @@ -120,7 +120,7 @@ def llm( @pytest.fixture(scope="module") def simple_llm() -> LLM: llm = LLM( - model=os.path.join(llm_models_root(), "llama-models-v2", "TinyLlama-1.1B-Chat-v1.0"), + model=os.path.join(llm_models_root(), "llama-3.1-model", "Llama-3.1-8B-Instruct"), max_batch_size=8, kv_cache_config=global_kvcache_config_prompt_logprobs, ) @@ -833,7 +833,7 @@ def test_processed_logprobs_e2e(logprobs_k: int, simple_llm: LLM): @force_ampere @pytest.mark.gpu2 def test_logprobs_match_hf_tp2(): - model_path = os.path.join(llm_models_root(), "llama-models-v2", "TinyLlama-1.1B-Chat-v1.0") + model_path = os.path.join(llm_models_root(), "llama-3.1-model", "Llama-3.1-8B-Instruct") llm = LLM( model=model_path, tensor_parallel_size=2, @@ -889,7 +889,7 @@ def test_logprobs_pp2(): Without the fix, logprobs length = 2N-1 instead of N due to duplication in the PP ring broadcast diff mechanism. """ - model_path = os.path.join(llm_models_root(), "llama-models-v2", "TinyLlama-1.1B-Chat-v1.0") + model_path = os.path.join(llm_models_root(), "llama-3.1-model", "Llama-3.1-8B-Instruct") max_tokens = 16 llm = LLM( model=model_path, diff --git a/tests/unittest/_torch/sampler/test_penalties_e2e.py b/tests/unittest/_torch/sampler/test_penalties_e2e.py index 553aa51717c4..99904ad60749 100644 --- a/tests/unittest/_torch/sampler/test_penalties_e2e.py +++ b/tests/unittest/_torch/sampler/test_penalties_e2e.py @@ -30,7 +30,7 @@ @pytest.fixture(scope="module") def model_path() -> Path: - return llm_models_root() / "llama-models-v2/TinyLlama-1.1B-Chat-v1.0" + return llm_models_root() / "llama-3.1-model/Llama-3.1-8B-Instruct" @dataclass(frozen=True) diff --git a/tests/unittest/_torch/sampler/test_trtllm_sampler.py b/tests/unittest/_torch/sampler/test_trtllm_sampler.py index 032a7bc21697..201b60759d16 100644 --- a/tests/unittest/_torch/sampler/test_trtllm_sampler.py +++ b/tests/unittest/_torch/sampler/test_trtllm_sampler.py @@ -9,7 +9,7 @@ @pytest.fixture(scope="module") def model_path(): - return llm_models_root() / "llama-models-v2/TinyLlama-1.1B-Chat-v1.0" + return llm_models_root() / "llama-3.1-model/Llama-3.1-8B-Instruct" def _create_llm_base(model_dir, enable_trtllm_sampler): diff --git a/tests/unittest/auto_deploy/singlegpu/smoke/test_ad_build_small_single.py b/tests/unittest/auto_deploy/singlegpu/smoke/test_ad_build_small_single.py index 1aa413d265de..9cd9c5113810 100644 --- a/tests/unittest/auto_deploy/singlegpu/smoke/test_ad_build_small_single.py +++ b/tests/unittest/auto_deploy/singlegpu/smoke/test_ad_build_small_single.py @@ -162,27 +162,6 @@ def _check_ad_config(experiment_config: ExperimentConfig, llm_args: LlmArgs): }, }, ), - ( - "meta-llama/Llama-4-Scout-17B-16E-Instruct", - { - "transforms": { - "insert_cached_attention": {"backend": "flashinfer"}, - "compile_model": { - "backend": "torch-simple", - "piecewise_enabled": False, - }, - }, - }, - ), - ( - "meta-llama/Llama-4-Scout-17B-16E-Instruct", - { - "transforms": { - "transformers_replace_cached_attn": {"backend": "flashinfer"}, - }, - "mode": "transformers", - }, - ), ( "deepseek-ai/DeepSeek-V3", { diff --git a/tests/unittest/grpc/smg/test_smg.py b/tests/unittest/grpc/smg/test_smg.py index 8d6290acf281..155ad5533901 100644 --- a/tests/unittest/grpc/smg/test_smg.py +++ b/tests/unittest/grpc/smg/test_smg.py @@ -325,13 +325,13 @@ def test_health_check_messages(self): def test_model_info_response(self): """Test GetModelInfoResponse message.""" response = pb2.GetModelInfoResponse( - model_id="meta-llama/Meta-Llama-3-8B", + model_id="meta-llama/Llama-3.1-8B-Instruct", max_input_len=4096, max_seq_len=8192, vocab_size=32000, ) - assert response.model_id == "meta-llama/Meta-Llama-3-8B" + assert response.model_id == "meta-llama/Llama-3.1-8B-Instruct" assert response.max_input_len == 4096 assert response.max_seq_len == 8192 assert response.vocab_size == 32000 @@ -642,7 +642,7 @@ def test_missing_tokenized_input(self): # End-to-end gRPC service tests (with real model) # ============================================================================ -default_model_name = "llama-models-v2/TinyLlama-1.1B-Chat-v1.0" +default_model_name = "llama-3.1-model/Llama-3.1-8B-Instruct" def get_model_path(model_name): @@ -656,11 +656,8 @@ def get_model_path(model_name): def grpc_service(): """Create a real LLM, request manager, and servicer for e2e testing. - Uses TinyLlama-1.1B for minimal GPU resource usage. - Shared across all tests in the class; class scope (not module) so the - LLM is shut down and its GPU memory released before the multimodal - class below creates its own LLM — with module scope both models are - alive at once and the second one OOMs on A10. + Uses Llama-3.1-8B-Instruct for GPU resource usage. + Shared across all tests in this module. """ model_path = get_model_path(default_model_name) llm = LLM( @@ -707,7 +704,7 @@ class TestGrpcServiceEndToEnd: """End-to-end tests for the gRPC service flow. Tests the full pipeline: gRPC request -> servicer -> request manager -> LLM -> response. - Uses TinyLlama-1.1B for minimal GPU resource usage. + Uses Llama-3.1-8B-Instruct for GPU resource usage. """ def test_generate_non_streaming(self, grpc_service): diff --git a/tests/unittest/llmapi/apps/_test_openai_lora.py b/tests/unittest/llmapi/apps/_test_openai_lora.py index 8e624122428c..9c0f09439daf 100644 --- a/tests/unittest/llmapi/apps/_test_openai_lora.py +++ b/tests/unittest/llmapi/apps/_test_openai_lora.py @@ -1,111 +1,3 @@ -import os -import tempfile -from dataclasses import asdict -from typing import List, Optional - -import openai import pytest -import yaml - -from tensorrt_llm.executor.request import LoRARequest - -from ..test_llm import get_model_path -from .openai_server import RemoteOpenAIServer pytestmark = pytest.mark.threadleak(enabled=False) - - -@pytest.fixture(scope="module", ids=["llama-models/llama-7b-hf"]) -def model_name() -> str: - return "llama-models/llama-7b-hf" - - -@pytest.fixture(scope="module") -def lora_adapter_names() -> List[Optional[str]]: - return [ - None, "llama-models/luotuo-lora-7b-0.1", - "llama-models/Japanese-Alpaca-LoRA-7b-v0" - ] - - -@pytest.fixture(scope="module") -def temp_extra_llm_api_options_file(): - temp_dir = tempfile.gettempdir() - temp_file_path = os.path.join(temp_dir, "extra_llm_api_options.yaml") - try: - extra_llm_api_options_dict = { - "lora_config": { - "lora_target_modules": ['attn_q', 'attn_k', 'attn_v'], - "max_lora_rank": 8, - "max_loras": 4, - "max_cpu_loras": 4, - }, - # Disable CUDA graph - # TODO: remove this once we have a proper fix for CUDA graph in LoRA - "cuda_graph_config": None - } - - with open(temp_file_path, 'w') as f: - yaml.dump(extra_llm_api_options_dict, f) - - yield temp_file_path - finally: - if os.path.exists(temp_file_path): - os.remove(temp_file_path) - - -@pytest.fixture(scope="module") -def server(model_name: str, - temp_extra_llm_api_options_file: str) -> RemoteOpenAIServer: - model_path = get_model_path(model_name) - args = [] - args.extend(["--backend", "pytorch"]) - args.extend(["--extra_llm_api_options", temp_extra_llm_api_options_file]) - with RemoteOpenAIServer(model_path, args) as remote_server: - yield remote_server - - -@pytest.fixture(scope="module") -def client(server: RemoteOpenAIServer) -> openai.OpenAI: - return server.get_client() - - -def test_lora(client: openai.OpenAI, model_name: str, - lora_adapter_names: List[str]): - prompts = [ - "美国的首都在哪里? \n答案:", - "美国的首都在哪里? \n答案:", - "美国的首都在哪里? \n答案:", - "アメリカ合衆国の首都はどこですか? \n答え:", - "アメリカ合衆国の首都はどこですか? \n答え:", - "アメリカ合衆国の首都はどこですか? \n答え:", - ] - references = [ - "沃尔玛\n\n## 新闻\n\n* ", - "美国的首都是华盛顿。\n\n美国的", - "纽约\n\n### カンファレンスの", - "Washington, D.C.\nWashington, D.C. is the capital of the United", - "华盛顿。\n\n英国の首都是什", - "ワシントン\nQ1. アメリカ合衆国", - ] - - for prompt, reference, lora_adapter_name in zip(prompts, references, - lora_adapter_names * 2): - extra_body = {} - if lora_adapter_name is not None: - lora_req = LoRARequest(lora_adapter_name, - lora_adapter_names.index(lora_adapter_name), - get_model_path(lora_adapter_name)) - extra_body["lora_request"] = asdict(lora_req) - - response = client.completions.create( - model=model_name, - prompt=prompt, - max_tokens=20, - extra_body=extra_body, - ) - # lora output is not deterministic, so do not check if match with reference - # TODO: need to fix this - print(f"response: {response.choices[0].text}") - print(f"reference: {reference}") - # assert similar(response.choices[0].text, reference) diff --git a/tests/unittest/llmapi/apps/_test_openai_prometheus.py b/tests/unittest/llmapi/apps/_test_openai_prometheus.py index a113274ad0c2..a3b4773d5d9d 100644 --- a/tests/unittest/llmapi/apps/_test_openai_prometheus.py +++ b/tests/unittest/llmapi/apps/_test_openai_prometheus.py @@ -25,7 +25,6 @@ import pytest import yaml -from ..test_llm import get_model_path from .openai_server import RemoteOpenAIServer # Configure logging @@ -33,12 +32,6 @@ logger = logging.getLogger(__name__) -@pytest.fixture(scope="module", ids=["TinyLlama-1.1B-Chat"]) -def model_name(): - """Return the HuggingFace model path used for all tests in this module.""" - return "llama-models-v2/TinyLlama-1.1B-Chat-v1.0" - - @pytest.fixture(scope="module") def temp_extra_llm_api_options_file(request): """Create a temporary YAML file with extra LLM API options for metrics collection.""" @@ -63,19 +56,6 @@ def temp_extra_llm_api_options_file(request): os.remove(temp_file_path) -@pytest.fixture(scope="module") -def server(model_name: str, - temp_extra_llm_api_options_file: str) -> RemoteOpenAIServer: - """Start a RemoteOpenAIServer with the PyTorch backend and metrics enabled.""" - model_path = get_model_path(model_name) - args = ["--backend", "pytorch", "--tp_size", "1"] - args.extend(["--extra_llm_api_options", temp_extra_llm_api_options_file]) - logger.info(f"Starting server, model: {model_name}, args: {args}") - with RemoteOpenAIServer(model_path, args) as remote_server: - yield remote_server - logger.info("Tests completed, shutting down server") - - def _parse_prometheus_sample(data: str, metric_name: str) -> float | None: """Parse Prometheus exposition text and return the sample value for a metric. diff --git a/tests/unittest/llmapi/apps/_test_trtllm_serve_example.py b/tests/unittest/llmapi/apps/_test_trtllm_serve_example.py index 4b75e4c71ff1..d9514a238699 100644 --- a/tests/unittest/llmapi/apps/_test_trtllm_serve_example.py +++ b/tests/unittest/llmapi/apps/_test_trtllm_serve_example.py @@ -1,80 +1,9 @@ -import json import os -import subprocess -import tempfile import pytest -import yaml - -from ..test_llm import get_model_path -from .openai_server import RemoteOpenAIServer - - -@pytest.fixture(scope="module", ids=["TinyLlama-1.1B-Chat"]) -def model_name(): - return "llama-models-v2/TinyLlama-1.1B-Chat-v1.0" - - -@pytest.fixture(scope="module") -def temp_extra_llm_api_options_file(): - temp_dir = tempfile.gettempdir() - temp_file_path = os.path.join(temp_dir, "extra_llm_api_options.yaml") - try: - extra_llm_api_options_dict = {"guided_decoding_backend": "xgrammar"} - with open(temp_file_path, 'w') as f: - yaml.dump(extra_llm_api_options_dict, f) - - yield temp_file_path - finally: - if os.path.exists(temp_file_path): - os.remove(temp_file_path) - - -@pytest.fixture(scope="module") -def server(model_name: str, temp_extra_llm_api_options_file: str): - model_path = get_model_path(model_name) - # fix port to facilitate concise trtllm-serve examples - args = ["--extra_llm_api_options", temp_extra_llm_api_options_file] - with RemoteOpenAIServer(model_path, args, port=8000) as remote_server: - yield remote_server @pytest.fixture(scope="module") def example_root(): llm_root = os.getenv("LLM_ROOT") return os.path.join(llm_root, "examples", "serve") - - -@pytest.mark.parametrize( - "exe, script", [("python3", "openai_chat_client.py"), - ("python3", "openai_completion_client.py"), - ("python3", "openai_completion_client_json_schema.py"), - ("python3", "openai_responses_client.py"), - ("bash", "curl_chat_client.sh"), - ("bash", "curl_completion_client.sh"), - ("bash", "aiperf_client.sh"), - ("bash", "curl_responses_client.sh")]) -def test_trtllm_serve_examples(exe: str, script: str, model_name: str, - server: RemoteOpenAIServer, example_root: str): - client_script = os.path.join(example_root, script) - # CalledProcessError will be raised if any errors occur - custom_env = os.environ.copy() - if script.startswith("aiperf"): - custom_env[""] = get_model_path(model_name) - result = subprocess.run([exe, client_script], - stdout=subprocess.PIPE, - stderr=subprocess.PIPE, - text=True, - check=True, - env=custom_env) - if script.startswith("curl"): - # For curl scripts, we expect a JSON response - result_stdout = result.stdout.strip() - try: - data = json.loads(result_stdout) - assert "code" not in data or data[ - "code"] == 200, f"Unexpected response: {data}" - except json.JSONDecodeError as e: - pytest.fail( - f"Failed to parse JSON response from {script}: {e}\nStdout: {result_stdout}\nStderr: {result.stderr}" - ) diff --git a/tests/unittest/llmapi/apps/_test_trtllm_serve_top_logprobs.py b/tests/unittest/llmapi/apps/_test_trtllm_serve_top_logprobs.py index 6c0d023b4726..51347e08e983 100644 --- a/tests/unittest/llmapi/apps/_test_trtllm_serve_top_logprobs.py +++ b/tests/unittest/llmapi/apps/_test_trtllm_serve_top_logprobs.py @@ -1,98 +1,8 @@ -import openai import pytest -from ..test_llm import get_model_path -from .openai_server import RemoteOpenAIServer - pytestmark = pytest.mark.threadleak(enabled=False) -@pytest.fixture(scope="module", ids=["TinyLlama-1.1B-Chat"]) -def model_name(): - return "llama-models-v2/TinyLlama-1.1B-Chat-v1.0" - - @pytest.fixture(scope="module", params=["pytorch"]) def backend(request): return request.param - - -@pytest.fixture(scope="module") -def server(model_name: str, backend: str): - model_path = get_model_path(model_name) - args = ["--backend", f"{backend}"] - with RemoteOpenAIServer(model_path, args) as remote_server: - yield remote_server - - -@pytest.fixture(scope="module") -def async_client(server: RemoteOpenAIServer): - return server.get_async_client() - - -@pytest.mark.asyncio(loop_scope="module") -async def test_chat_completion_top5_logprobs(async_client: openai.AsyncOpenAI, - model_name: str): - messages = [{ - "role": "system", - "content": "You are a helpful assistant." - }, { - "role": "user", - "content": "What is the capital of France?" - }] - - # Test top_logprobs - chat_completion = await async_client.chat.completions.create( - model=model_name, - messages=messages, # type: ignore[arg-type] - max_completion_tokens=10, - temperature=0.0, - logprobs=True, - top_logprobs=5, - extra_body={ - "ignore_eos": True, - }) - logprobs = chat_completion.choices[0].logprobs - assert logprobs is not None and logprobs.content is not None - assert len(logprobs.content) == 10 - for logprob_content in logprobs.content: - assert logprob_content.token is not None - assert logprob_content.logprob is not None - assert logprob_content.bytes is not None - assert logprob_content.top_logprobs is not None - assert len(logprob_content.top_logprobs) == 5 - - -@pytest.mark.asyncio(loop_scope="module") -async def test_completion_top5_logprobs(async_client: openai.AsyncOpenAI, - model_name: str): - prompt = "Hello, my name is" - - completion = await async_client.completions.create(model=model_name, - prompt=prompt, - max_tokens=5, - temperature=0.0, - logprobs=5, - extra_body={ - "ignore_eos": True, - }) - - choice = completion.choices[0] - logprobs = choice.logprobs - assert logprobs is not None - assert logprobs.tokens is not None - assert logprobs.token_logprobs is not None - assert logprobs.top_logprobs is not None - - assert len(logprobs.tokens) == len(logprobs.token_logprobs) == len( - logprobs.top_logprobs) - assert len(logprobs.tokens) > 0 - - for token, token_logprob, token_top_logprobs in zip(logprobs.tokens, - logprobs.token_logprobs, - logprobs.top_logprobs): - assert token is not None - assert token_logprob is not None - assert token_logprob <= 0 - assert token_top_logprobs is not None - assert len(token_top_logprobs) == 5 diff --git a/tests/unittest/llmapi/lora_test_utils.py b/tests/unittest/llmapi/lora_test_utils.py index 2c4b97825481..089ec0bc4437 100644 --- a/tests/unittest/llmapi/lora_test_utils.py +++ b/tests/unittest/llmapi/lora_test_utils.py @@ -3,12 +3,11 @@ import tempfile from dataclasses import asdict, dataclass from pathlib import Path -from typing import List, Optional, OrderedDict, Tuple, Type, Union +from typing import List, Optional, Tuple, Type, Union import pytest import torch from utils.llm_data import llm_models_root -from utils.util import duplicate_list_to_length, flatten_list, similar from tensorrt_llm import SamplingParams from tensorrt_llm._torch.peft.lora.cuda_graph_lora_params import \ @@ -75,114 +74,6 @@ def check_phi3_lora_fused_modules_output_tp2_identical_to_tp1( assert outputs_tp1 == outputs_tp2 -def check_llama_7b_multi_unique_lora_adapters_from_request( - lora_adapter_count_per_call: List[int], repeat_calls: int, - repeats_per_call: int, llm_class: Type[BaseLLM], **llm_kwargs): - """Calls llm.generate s.t. for each C in lora_adapter_count_per_call, llm.generate is called with C requests - repeated 'repeats_per_call' times, where each request is configured with a unique LoRA adapter ID. - This entire process is done in a loop 'repeats_per_call' times with the same requests. - Asserts the output of each llm.generate call is similar to the expected. - """ # noqa: D205 - total_lora_adapters = sum(lora_adapter_count_per_call) - hf_model_dir = f"{llm_models_root()}/llama-models/llama-7b-hf" - hf_lora_dirs = [ - f"{llm_models_root()}/llama-models/luotuo-lora-7b-0.1", - f"{llm_models_root()}/llama-models/Japanese-Alpaca-LoRA-7b-v0" - ] - # Each prompt should have a reference for every LoRA adapter dir (in the same order as in hf_lora_dirs) - prompt_to_references = OrderedDict({ - "美国的首都在哪里? \n答案:": [ - "美国的首都是华盛顿。\n\n美国的", - "纽约\n\n### カンファレンスの", - ], - "アメリカ合衆国の首都はどこですか? \n答え:": [ - "华盛顿。\n\n英国の首都是什", - "ワシントン\nQ1. アメリカ合衆国", - ], - }) - - prompts_to_generate = duplicate_list_to_length( - flatten_list([[prompt] * len(hf_lora_dirs) - for prompt in prompt_to_references.keys()]), - total_lora_adapters) - references = duplicate_list_to_length( - flatten_list(list(prompt_to_references.values())), total_lora_adapters) - lora_requests = [ - LoRARequest(str(i), i, hf_lora_dirs[i % len(hf_lora_dirs)]) - for i in range(total_lora_adapters) - ] - llm = llm_class(hf_model_dir, **llm_kwargs) - - # Perform repeats of the same requests to test reuse and reload of adapters previously unloaded from cache - try: - for _ in range(repeat_calls): - last_idx = 0 - for adapter_count in lora_adapter_count_per_call: - sampling_params = SamplingParams(max_tokens=20) - outputs = llm.generate( - prompts_to_generate[last_idx:last_idx + adapter_count] * - repeats_per_call, - sampling_params, - lora_request=lora_requests[last_idx:last_idx + - adapter_count] * - repeats_per_call) - for output, ref in zip( - outputs, references[last_idx:last_idx + adapter_count] * - repeats_per_call): - assert similar(output.outputs[0].text, ref) - last_idx += adapter_count - finally: - llm.shutdown() - - -def check_llama_7b_multi_lora_from_request_test_harness( - llm_class: Type[BaseLLM], **llm_kwargs) -> None: - hf_model_dir = f"{llm_models_root()}/llama-models/llama-7b-hf" - hf_lora_dir1 = f"{llm_models_root()}/llama-models/luotuo-lora-7b-0.1" - hf_lora_dir2 = f"{llm_models_root()}/llama-models/Japanese-Alpaca-LoRA-7b-v0" - prompts = [ - "美国的首都在哪里? \n答案:", - "美国的首都在哪里? \n答案:", - "美国的首都在哪里? \n答案:", - "アメリカ合衆国の首都はどこですか? \n答え:", - "アメリカ合衆国の首都はどこですか? \n答え:", - "アメリカ合衆国の首都はどこですか? \n答え:", - ] - references = [ - "沃尔玛\n\n## 新闻\n\n* ", - "美国的首都是华盛顿。\n\n美国的", - "纽约\n\n### カンファレンスの", - "Washington, D.C.\nWashington, D.C. is the capital of the United", - "华盛顿。\n\n英国の首都是什", - "ワシントン\nQ1. アメリカ合衆国", - ] - key_words = [ - "沃尔玛", - "华盛顿", - "纽约", - "Washington", - "华盛顿", - "ワシントン", - ] - lora_req1 = LoRARequest("luotuo", 1, hf_lora_dir1) - lora_req2 = LoRARequest("Japanese", 2, hf_lora_dir2) - sampling_params = SamplingParams(max_tokens=20) - - llm = llm_class(hf_model_dir, **llm_kwargs) - try: - outputs = llm.generate(prompts, - sampling_params, - lora_request=[ - None, lora_req1, lora_req2, None, lora_req1, - lora_req2 - ]) - finally: - llm.shutdown() - for output, ref, key_word in zip(outputs, references, key_words): - assert similar(output.outputs[0].text, - ref) or key_word in output.outputs[0].text - - def create_mock_nemo_lora_checkpoint( lora_dir: Path, hidden_size: int = 4096, @@ -245,9 +136,11 @@ def create_mock_nemo_lora_checkpoint( qkv_output_dim = hidden_size + 2 * kv_hidden_size # NOTE: - # for seed=42, and coefficient=0.02, the expected outputs are hardcoded - # in the test `test_llm_pytorch.py::test_gqa_nemo_lora`. - # Therefore changing "WEIGHTS_COEFFICIENT" or the seed will break the test. + # test_llm_pytorch.py::test_gqa_nemo_lora asserts that the LoRA adapter + # changes the generated text relative to a no-LoRA baseline (difference-based + # assertion — no output strings are hardcoded). Both seed=42 and + # WEIGHTS_COEFFICIENT affect whether the adapter is large enough to shift + # the output, so changes to either may cause that test to fail. WEIGHTS_COEFFICIENT = 0.02 for layer_idx in range(num_layers): key_prefix = f"model.layers.{layer_idx}.self_attention.adapter_layer.lora_kqv_adapter" diff --git a/tests/unittest/llmapi/test_async_llm.py b/tests/unittest/llmapi/test_async_llm.py index 9468eedaa6d5..8604781389b7 100644 --- a/tests/unittest/llmapi/test_async_llm.py +++ b/tests/unittest/llmapi/test_async_llm.py @@ -16,7 +16,7 @@ @pytest.mark.ray @pytest.mark.asyncio async def test_async_llm_awaitable(): - llama_model_path = str(llm_models_root() / "llama-models-v2/TinyLlama-1.1B-Chat-v1.0") + llama_model_path = str(llm_models_root() / "llama-3.1-model/Llama-3.1-8B-Instruct") kv_cache_config = KvCacheConfig(enable_block_reuse=False) prompt = "The future of AI is" @@ -41,7 +41,7 @@ async def test_async_llm_awaitable(): @pytest.mark.asyncio @pytest.mark.parametrize("num_cycles", [3], ids=lambda x: f"{x}_cycle") async def test_async_llm_release_resume(process_gpu_memory_info_available, num_cycles): - llama_model_path = str(llm_models_root() / "llama-models-v2/TinyLlama-1.1B-Chat-v1.0") + llama_model_path = str(llm_models_root() / "llama-3.1-model/Llama-3.1-8B-Instruct") kv_cache_config = KvCacheConfig(enable_block_reuse=False, max_tokens=4096) prompt = "The future of AI is" @@ -113,9 +113,7 @@ async def test_async_llm_placement_api(setup_ray_cluster, monkeypatch): print(f"Placement group ready with bundles {pg.bundle_specs}") llm = await AsyncLLM( - model=os.path.join( - str(llm_models_root()), "llama-models-v2", "TinyLlama-1.1B-Chat-v1.0" - ), + model=os.path.join(str(llm_models_root()), "llama-3.1-model", "Llama-3.1-8B-Instruct"), kv_cache_config=KvCacheConfig(free_gpu_memory_fraction=0.1), tensor_parallel_size=tp_size, placement_groups=[pg], @@ -141,7 +139,7 @@ async def test_async_llm_placement_api(setup_ray_cluster, monkeypatch): @pytest.mark.ray @pytest.mark.asyncio async def test_async_llm_reset_prefix_cache(): - llama_model_path = str(llm_models_root() / "llama-models-v2/TinyLlama-1.1B-Chat-v1.0") + llama_model_path = str(llm_models_root() / "llama-3.1-model/Llama-3.1-8B-Instruct") kv_cache_config = KvCacheConfig(enable_block_reuse=True) prompt = "The future of AI is " * 20 sampling_params = SamplingParams(temperature=0, max_tokens=5, return_perf_metrics=True) @@ -182,7 +180,7 @@ async def test_async_llm_reset_prefix_cache(): @pytest.mark.ray @pytest.mark.asyncio async def test_async_llm_pause_resume(): - llama_model_path = str(llm_models_root() / "llama-models-v2/TinyLlama-1.1B-Chat-v1.0") + llama_model_path = str(llm_models_root() / "llama-3.1-model/Llama-3.1-8B-Instruct") prompt = "The future of AI is" sampling_params = SamplingParams(temperature=0, max_tokens=10) @@ -209,7 +207,7 @@ async def test_async_llm_pause_resume(): @pytest.mark.ray @pytest.mark.asyncio async def test_async_llm_pause_aborts_inflight(): - llama_model_path = str(llm_models_root() / "llama-models-v2/TinyLlama-1.1B-Chat-v1.0") + llama_model_path = str(llm_models_root() / "llama-3.1-model/Llama-3.1-8B-Instruct") prompt = "The future of AI is" inflight_params = SamplingParams(temperature=0, max_tokens=512) normal_params = SamplingParams(temperature=0, max_tokens=10) diff --git a/tests/unittest/llmapi/test_executor.py b/tests/unittest/llmapi/test_executor.py index 5aaa3433794a..b508f241520f 100644 --- a/tests/unittest/llmapi/test_executor.py +++ b/tests/unittest/llmapi/test_executor.py @@ -191,7 +191,7 @@ def test_result_completes_within_timeout(): def test_DetokenizedGenerationResultBase(): sampling_params = SamplingParams(max_tokens=4) - model_path = llm_models_root() / "llama-models-v2/TinyLlama-1.1B-Chat-v1.0" + model_path = llm_models_root() / "llama-3.1-model/Llama-3.1-8B-Instruct" tokenizer = TransformersTokenizer.from_pretrained(model_path) result = DetokenizedGenerationResultBase( id=2, @@ -408,7 +408,7 @@ def test_ResponsePostprocessWorker(): fut = pool.submit( ResponsePostprocessWorker_worker_task, input_pipe.address, out_pipe.address, - str(llm_models_root() / "llama-models-v2/TinyLlama-1.1B-Chat-v1.0")) + str(llm_models_root() / "llama-3.1-model/Llama-3.1-8B-Instruct")) inputs = [ Input(rsp=create_rsp(123), @@ -503,7 +503,7 @@ def test_PostprocWorker_disaggregated_params(): fut = pool.submit( ResponsePostprocessWorker_worker_task, input_pipe.address, out_pipe.address, - str(llm_models_root() / "llama-models-v2/TinyLlama-1.1B-Chat-v1.0")) + str(llm_models_root() / "llama-3.1-model/Llama-3.1-8B-Instruct")) disagg_params = DisaggregatedParams( request_type="generation_only", diff --git a/tests/unittest/llmapi/test_llm.py b/tests/unittest/llmapi/test_llm.py index 660bac8c7d67..6a287a910faf 100644 --- a/tests/unittest/llmapi/test_llm.py +++ b/tests/unittest/llmapi/test_llm.py @@ -37,10 +37,6 @@ # isort: on -# The unittests are based on the tiny-llama, which is fast to build and run. -# There are other tests based on llama-7B model, such as the end-to-end tests in test_e2e.py, and parallel tests in -# test_llm_multi_gpu.py. - pytestmark = pytest.mark.threadleak(enabled=False) @@ -124,7 +120,7 @@ def llm_check_output(llm: LLM, stop_reasons=stop_reasons) -default_model_name = "llama-models-v2/TinyLlama-1.1B-Chat-v1.0" +default_model_name = "llama-3.1-model/Llama-3.1-8B-Instruct" mixtral_model_name = "Mixtral-8x7B-v0.1" llama_model_path = get_model_path(default_model_name) @@ -383,41 +379,6 @@ async def main(): test_non_streaming_usage_wait() -@pytest.mark.parametrize("chunked", [True, False]) -@pytest.mark.part0 -@pytest.mark.mpi_ray_parity -def test_llm_generate_async_with_stream_interval(chunked): - model_path = get_model_path('llama-models-v2/llama-v2-7b-hf') - max_num_tokens = 256 - with LLM(model_path, - max_num_tokens=max_num_tokens, - stream_interval=4, - enable_chunked_prefill=chunked) as llm: - sampling_params = SamplingParams(max_tokens=13, - ignore_eos=True, - detokenize=False) - step = 0 - last_step_len = 0 - prompt = "The capital of France is " - if chunked: - prompt = prompt * max_num_tokens - for output in llm.generate_async(prompt, - sampling_params=sampling_params, - streaming=True): - current_step_len = len(output.outputs[0].token_ids) - # The output lens of each step need to be [1, 3, 4, 4, 1] - if step == 0: - assert current_step_len == 1 - elif step == 1: - assert current_step_len - last_step_len == 3 - elif step == 2 or step == 3: - assert current_step_len - last_step_len == 4 - else: - assert current_step_len - last_step_len == 1 - step += 1 - last_step_len = current_step_len - - @pytest.mark.part0 def test_parallel_config(): config = _ParallelConfig() diff --git a/tests/unittest/llmapi/test_llm_kv_cache_events.py b/tests/unittest/llmapi/test_llm_kv_cache_events.py index d2447c1aea2f..e442fada9e47 100644 --- a/tests/unittest/llmapi/test_llm_kv_cache_events.py +++ b/tests/unittest/llmapi/test_llm_kv_cache_events.py @@ -30,7 +30,7 @@ from .test_llm import get_model_path -default_model_name = "llama-models-v2/TinyLlama-1.1B-Chat-v1.0" +default_model_name = "llama-3.1-model/Llama-3.1-8B-Instruct" llama_model_path = get_model_path(default_model_name) global_kvcache_config = KvCacheConfig(free_gpu_memory_fraction=0.4, event_buffer_max_size=1024, diff --git a/tests/unittest/llmapi/test_llm_multi_gpu_pytorch.py b/tests/unittest/llmapi/test_llm_multi_gpu_pytorch.py index 0e15b38b8b95..5e3bf32d0d7f 100644 --- a/tests/unittest/llmapi/test_llm_multi_gpu_pytorch.py +++ b/tests/unittest/llmapi/test_llm_multi_gpu_pytorch.py @@ -4,19 +4,15 @@ from tensorrt_llm import LLM from tensorrt_llm.executor.rpc_proxy import GenerationExecutorRpcProxy from tensorrt_llm.llmapi import KvCacheConfig -from tensorrt_llm.lora_helper import LoraConfig from tensorrt_llm.sampling_params import SamplingParams from .lora_test_utils import ( - check_llama_7b_multi_lora_from_request_test_harness, check_phi3_lora_fused_modules_output_tp2_identical_to_tp1, test_lora_with_and_without_cuda_graph) from .test_llm import (_test_llm_capture_request_error, llama_model_path, llm_get_stats_async_test_harness, llm_get_stats_test_harness, - llm_return_logprobs_test_harness, - tinyllama_logits_processor_test_harness) -from .test_llm_pytorch import llama_7b_lora_from_dir_test_harness + llm_return_logprobs_test_harness) global_kv_cache_config = KvCacheConfig(free_gpu_memory_fraction=0.4) @@ -26,47 +22,6 @@ def test_llm_capture_request_error(): _test_llm_capture_request_error(pytorch_backend=True, tp_size=2) -@pytest.mark.gpu4 -def test_tinyllama_logits_processor_tp2pp2(): - tinyllama_logits_processor_test_harness(backend="pytorch", - tensor_parallel_size=2, - pipeline_parallel_size=2) - - -@pytest.mark.gpu2 -@pytest.mark.part0 -@pytest.mark.parametrize("tp_size, pp_size", [(1, 2), (2, 1)]) -def test_tinyllama_logits_processor_2gpu(tp_size: int, pp_size: int): - tinyllama_logits_processor_test_harness(backend="pytorch", - tensor_parallel_size=tp_size, - pipeline_parallel_size=pp_size) - - -@pytest.mark.gpu2 -def test_llama_7b_lora_tp2(): - llama_7b_lora_from_dir_test_harness(tensor_parallel_size=2, - kv_cache_config=global_kv_cache_config) - - -@pytest.mark.gpu4 -@skip_ray # https://nvbugs/5682551 -@test_lora_with_and_without_cuda_graph -def test_llama_7b_multi_lora_tp4(cuda_graph_config): - # For LoRA checkpoints without finetuned embedding and lm_head, we can either: - # (1) specify lora_target_modules, or - # (2) provide a lora_dir to infer the lora_target_modules. - lora_config = LoraConfig(lora_target_modules=['attn_q', 'attn_k', 'attn_v'], - max_lora_rank=8, - max_loras=1, - max_cpu_loras=8) - check_llama_7b_multi_lora_from_request_test_harness( - LLM, - lora_config=lora_config, - tensor_parallel_size=4, - kv_cache_config=global_kv_cache_config, - cuda_graph_config=cuda_graph_config) - - @skip_ray # https://nvbugs/5727075 @pytest.mark.gpu2 @test_lora_with_and_without_cuda_graph diff --git a/tests/unittest/llmapi/test_llm_pytorch.py b/tests/unittest/llmapi/test_llm_pytorch.py index bce68e10531e..f324bbb6a0c9 100644 --- a/tests/unittest/llmapi/test_llm_pytorch.py +++ b/tests/unittest/llmapi/test_llm_pytorch.py @@ -16,18 +16,17 @@ from tensorrt_llm.executor import GenerationExecutorWorker, RequestError from tensorrt_llm.executor.rpc_proxy import GenerationExecutorRpcProxy from tensorrt_llm.llmapi import CacheTransceiverConfig, KvCacheConfig -from tensorrt_llm.llmapi.llm_args import (NGramDecodingConfig, PeftCacheConfig, - SchedulerConfig, WaitingQueuePolicy) +from tensorrt_llm.llmapi.llm_args import (NGramDecodingConfig, SchedulerConfig, + WaitingQueuePolicy) from tensorrt_llm.llmapi.tokenizer import TransformersTokenizer from tensorrt_llm.metrics import MetricNames from tensorrt_llm.sampling_params import SamplingParams # isort: off -from .lora_test_utils import ( - check_llama_7b_multi_lora_from_request_test_harness, - check_llama_7b_multi_unique_lora_adapters_from_request, - create_mock_nemo_lora_checkpoint, compare_cuda_graph_lora_params_filler, - CUDAGraphLoRATestParams, test_lora_with_and_without_cuda_graph) +from .lora_test_utils import (create_mock_nemo_lora_checkpoint, + compare_cuda_graph_lora_params_filler, + CUDAGraphLoRATestParams, + test_lora_with_and_without_cuda_graph) from .test_llm import (_test_llm_capture_request_error, get_model_path, global_kvcache_config, global_kvcache_config_no_reuse, llama_model_path, llm_get_stats_async_test_harness, @@ -39,8 +38,7 @@ tinyllama_logits_processor_test_harness) from utils.util import (force_ampere, similar, similarity_score, skip_fp8_pre_ada, skip_gpu_memory_less_than_40gb, - skip_gpu_memory_less_than_80gb, - skip_gpu_memory_less_than_138gb, skip_ray) + skip_gpu_memory_less_than_80gb, skip_ray) from utils.llm_data import llm_models_root from tensorrt_llm.lora_helper import LoraConfig from tensorrt_llm.executor.request import LoRARequest @@ -256,7 +254,7 @@ def test_llm_perf_metrics(): @pytest.mark.part3 @pytest.mark.parametrize("attn_backend", ["TRTLLM", "FLASHINFER"]) def test_llm_prefix_cache_reuse(attn_backend): - model_path = get_model_path("llama-models-v2/TinyLlama-1.1B-Chat-v1.0") + model_path = get_model_path("llama-3.1-model/Llama-3.1-8B-Instruct") prompt = "The future of AI is " * 20 sampling_params = SamplingParams(temperature=0, max_tokens=5, @@ -398,269 +396,6 @@ def test_lora_cuda_graph_params_filling_kernel_special_cases(): compare_cuda_graph_lora_params_filler(test_params6) -def llama_7b_lora_from_dir_test_harness(**llm_kwargs) -> None: - lora_config = LoraConfig( - lora_dir=[f"{llm_models_root()}/llama-models/luotuo-lora-7b-0.1"], - max_lora_rank=8, - max_loras=2, - max_cpu_loras=2) - llm = LLM(model=f"{llm_models_root()}/llama-models/llama-7b-hf", - lora_config=lora_config, - **llm_kwargs) - try: - prompts = [ - "美国的首都在哪里? \n答案:", - ] - references = [ - "美国的首都是华盛顿。\n\n美国的", - ] - sampling_params = SamplingParams(max_tokens=20) - lora_req = LoRARequest( - "task-0", 0, f"{llm_models_root()}/llama-models/luotuo-lora-7b-0.1") - lora_request = [lora_req] - - outputs = llm.generate(prompts, - sampling_params, - lora_request=lora_request) - assert similar(outputs[0].outputs[0].text, references[0]) - finally: - llm.shutdown() - - -@skip_gpu_memory_less_than_40gb -@pytest.mark.part0 -@test_lora_with_and_without_cuda_graph -@pytest.mark.parametrize("use_speculative", [True, False]) -def test_llama_7b_lora(cuda_graph_config, use_speculative): - llm_kwargs = { - "cuda_graph_config": - cuda_graph_config, - "speculative_config": - NGramDecodingConfig(max_draft_len=5) if use_speculative else None - } - llama_7b_lora_from_dir_test_harness(**llm_kwargs) - - -@skip_gpu_memory_less_than_40gb -@test_lora_with_and_without_cuda_graph -@pytest.mark.parametrize("use_speculative", [True, False]) -def test_llama_7b_lora_default_modules(cuda_graph_config, - use_speculative) -> None: - lora_config = LoraConfig(max_lora_rank=64, max_loras=2, max_cpu_loras=2) - - hf_model_dir = f"{llm_models_root()}/llama-models/llama-7b-hf" - - llm = LLM(model=hf_model_dir, - lora_config=lora_config, - speculative_config=NGramDecodingConfig( - max_draft_len=5) if use_speculative else None, - cuda_graph_config=cuda_graph_config) - - hf_lora_dir = f"{llm_models_root()}/llama-models/luotuo-lora-7b-0.1" - try: - prompts = [ - "美国的首都在哪里? \n答案:", - ] - references = [ - "美国的首都是华盛顿。\n\n美国的", - ] - sampling_params = SamplingParams(max_tokens=20, - add_special_tokens=False) - lora_req = LoRARequest("luotuo", 1, hf_lora_dir) - lora_request = [lora_req] - - outputs = llm.generate(prompts, - sampling_params, - lora_request=lora_request) - - assert similar(outputs[0].outputs[0].text, references[0]) - finally: - llm.shutdown() - - -def _check_llama_7b_multi_lora_evict_load_new_adapters( - lora_adapter_count_per_call: list[int], max_loras: int, - max_cpu_loras: int, repeat_calls: int, repeats_per_call: int, - **llm_kwargs): - # For LoRA checkpoints without finetuned embedding and lm_head, we can either: - # (1) specify lora_target_modules, or - # (2) provide a lora_dir to infer the lora_target_modules. - lora_config = LoraConfig(lora_target_modules=['attn_q', 'attn_k', 'attn_v'], - max_lora_rank=8, - max_loras=max_loras, - max_cpu_loras=max_cpu_loras) - check_llama_7b_multi_unique_lora_adapters_from_request( - lora_adapter_count_per_call, - repeat_calls, - repeats_per_call, - LLM, - lora_config=lora_config, - **llm_kwargs) - - -@skip_gpu_memory_less_than_40gb -@skip_ray # https://nvbugs/5682551 -@pytest.mark.part3 -@test_lora_with_and_without_cuda_graph -def test_llama_7b_multi_lora_evict_and_reload_lora_gpu_cache(cuda_graph_config): - """Test eviction and re-loading a previously evicted adapter from the LoRA GPU cache, within a single - llm.generate call, that's repeated twice. - """ # noqa: D205 - _check_llama_7b_multi_lora_evict_load_new_adapters( - lora_adapter_count_per_call=[2], - max_loras=1, - max_cpu_loras=2, - repeat_calls=2, - repeats_per_call=3, - cuda_graph_config=cuda_graph_config) - - -@skip_gpu_memory_less_than_40gb -@pytest.mark.part1 -@test_lora_with_and_without_cuda_graph -def test_llama_7b_multi_lora_evict_and_load_new_adapters_in_cpu_and_gpu_cache( - cuda_graph_config): - """Test eviction and loading of new adapters in the evicted space, over several llm.generate calls, with LoRA GPU - cache size < LoRA CPU cache size. - """ # noqa: D205 - _check_llama_7b_multi_lora_evict_load_new_adapters( - lora_adapter_count_per_call=[2, 2, 2], - max_loras=1, - max_cpu_loras=3, - repeat_calls=1, - repeats_per_call=1, - cuda_graph_config=cuda_graph_config) - - -@skip_gpu_memory_less_than_40gb -@pytest.mark.part0 -@test_lora_with_and_without_cuda_graph -def test_llama_7b_multi_lora_read_from_cache_after_insert(cuda_graph_config): - """Test that loading and then using the same adapters loaded in cache works.""" - _check_llama_7b_multi_lora_evict_load_new_adapters( - lora_adapter_count_per_call=[3], - max_loras=3, - max_cpu_loras=3, - repeat_calls=2, - repeats_per_call=1, - cuda_graph_config=cuda_graph_config) - - -@skip_gpu_memory_less_than_40gb -@pytest.mark.part3 -@test_lora_with_and_without_cuda_graph -def test_llama_7b_multi_lora_evict_and_reload_evicted_adapters_in_cpu_and_gpu_cache( - cuda_graph_config): - """Test eviction, reloading new adapters and reloading previously evicted adapters from the LoRA CPU cache & GPU - cache over multiple llm.generate call repeated twice (two calls with the same requests): - At the end of the 1st llm.generate call: - The LoRA caches should contain adapters 1, 2 and shouldn't contain adapter 0 (it should have been evicted). - So in the 2nd call, the worker should: - - Send req0 with adapter 0 weights (because it was previously evicted) - - Send the other two requests without their adapter weights as they're already in LoRA CPU cache - Then, handling of req0 that has weights but not in the cache should evict one of the other two adapters from - the cache, causing that evicted adapter's request to again load its weights from the file system, as they - aren't with the request and aren't in LoRA cache. - """ # noqa: D205 - _check_llama_7b_multi_lora_evict_load_new_adapters( - lora_adapter_count_per_call=[3], - max_loras=2, - max_cpu_loras=2, - repeat_calls=2, - repeats_per_call=1, - cuda_graph_config=cuda_graph_config) - - -@skip_gpu_memory_less_than_40gb -@pytest.mark.part2 -@test_lora_with_and_without_cuda_graph -def test_llama_7b_peft_cache_config_affects_peft_cache_size(cuda_graph_config): - """Tests that LLM arg of peft_cache_config affects the peft cache sizes. - - NOTE: The caller can't get the actual LoRA cache sizes, so we instead we - test that it fails when configured with a value too small to contain a - single adapter. - """ - # For LoRA checkpoints without finetuned embedding and lm_head, we can either: - # (1) specify lora_target_modules, or - # (2) provide a lora_dir to infer the lora_target_modules. - lora_config_no_cache_size_values = LoraConfig( - lora_target_modules=['attn_q', 'attn_k', 'attn_v'], max_lora_rank=8) - - # Test that too small PeftCacheConfig.host_cache_size causes failure - with pytest.raises(RuntimeError): - check_llama_7b_multi_lora_from_request_test_harness( - LLM, - lora_config=lora_config_no_cache_size_values, - peft_cache_config=PeftCacheConfig( - host_cache_size=1), # size in bytes - cuda_graph_config=cuda_graph_config) - - # Test that too small PeftCacheConfig.device_cache_percent causes failure - with pytest.raises(RuntimeError): - check_llama_7b_multi_lora_from_request_test_harness( - LLM, - lora_config=lora_config_no_cache_size_values, - peft_cache_config=PeftCacheConfig(device_cache_percent=0.0000001), - cuda_graph_config=cuda_graph_config) - - -@skip_ray # https://nvbugs/5682551 -@skip_gpu_memory_less_than_40gb -@pytest.mark.part1 -@test_lora_with_and_without_cuda_graph -def test_llama_7b_lora_config_overrides_peft_cache_config(cuda_graph_config): - """Tests that cache size args in lora_config LLM arg override the cache size - parameters in peft_cache_config LLM arg. - """ # noqa: D205 - check_llama_7b_multi_lora_from_request_test_harness( - LLM, - lora_config=LoraConfig( - lora_target_modules=['attn_q', 'attn_k', 'attn_v'], - max_lora_rank=8, - max_loras=2, - max_cpu_loras=2), - peft_cache_config=PeftCacheConfig( - host_cache_size=1, # size in bytes - device_cache_percent=0.0000001), - cuda_graph_config=cuda_graph_config) - - -@skip_gpu_memory_less_than_138gb -@pytest.mark.part1 -@test_lora_with_and_without_cuda_graph -def test_nemotron_nas_lora(cuda_graph_config) -> None: - lora_config = LoraConfig(lora_dir=[ - f"{llm_models_root()}/nemotron-nas/Llama-3_3-Nemotron-Super-49B-v1-lora-adapter_r64" - ], - max_lora_rank=64, - max_loras=1, - max_cpu_loras=1) - - llm = LLM( - model= - f"{llm_models_root()}/nemotron-nas/Llama-3_3-Nemotron-Super-49B-v1", - lora_config=lora_config, - cuda_graph_config=cuda_graph_config, - trust_remote_code=True) - - prompts = [ - "Hello, how are you?", - "Hello, how are you?", - ] - - sampling_params = SamplingParams(max_tokens=3, add_special_tokens=False) - lora_req = LoRARequest( - "task-0", 0, - f"{llm_models_root()}/nemotron-nas/Llama-3_3-Nemotron-Super-49B-v1-lora-adapter_r64" - ) - lora_request = [lora_req, None] - - outputs = llm.generate(prompts, sampling_params, lora_request=lora_request) - - assert similar(outputs[0].outputs[0].text, outputs[1].outputs[0].text) - - @skip_gpu_memory_less_than_80gb @pytest.mark.part0 @test_lora_with_and_without_cuda_graph @@ -930,22 +665,20 @@ def test_nemo_lora_unsupported_modules_validation(tmp_path): @pytest.mark.part1 @test_lora_with_and_without_cuda_graph def test_gqa_nemo_lora(tmp_path, cuda_graph_config): - """Test NeMo-format LoRA checkpoint loading and GQA support in TinyLlama. + """Test NeMo-format LoRA checkpoint loading and GQA support in Llama-3.1-8B-Instruct. This test verifies two properties: - 1. That a NeMo-format LoRA checkpoint with GQA (grouped query attention) can be loaded and applied to a TinyLlama model, - and that generation with this LoRA produces a deterministic, expected output for a fixed prompt and temperature=0.0. + 1. That a NeMo-format LoRA checkpoint with GQA (grouped query attention) can be loaded and applied. 2. That the LoRA weights have a significant effect: generating with LoRA produces a different output than generating without LoRA, confirming that the LoRA adapter is actually being applied. - The test uses a deterministic dummy LoRA checkpoint (seed=42) and checks both the positive (LoRA applied) and negative - (no LoRA) cases for output text. + The test uses a deterministic dummy LoRA checkpoint (seed=42). """ - # TinyLlama's exact GQA configuration - hidden_size = 2048 - num_layers = 22 + # Llama-3.1-8B-Instruct GQA configuration + hidden_size = 4096 + num_layers = 32 num_q_heads = 32 # Query attention heads - num_kv_heads = 4 # Key/Value heads (GQA) + num_kv_heads = 8 # Key/Value heads (GQA) lora_rank = 8 nemo_path = create_mock_nemo_lora_checkpoint( @@ -955,9 +688,8 @@ def test_gqa_nemo_lora(tmp_path, cuda_graph_config): lora_rank=lora_rank, num_attention_heads=num_q_heads, num_kv_heads=num_kv_heads, - seed=42, # NOTE: the seed=42 is important for the test to pass. + seed=42, ) - expected_lora_text_output = "Paris. The capital of France is Paris. The" test_prompts = ["The capital of France is"] sampling_params = SamplingParams(max_tokens=10, temperature=0.0) @@ -967,7 +699,7 @@ def test_gqa_nemo_lora(tmp_path, cuda_graph_config): max_lora_rank=lora_rank, ) - model_path = get_model_path("llama-models-v2/TinyLlama-1.1B-Chat-v1.0") + model_path = get_model_path("llama-3.1-model/Llama-3.1-8B-Instruct") llm = LLM( model=model_path, @@ -977,7 +709,7 @@ def test_gqa_nemo_lora(tmp_path, cuda_graph_config): ) try: - lora_req = LoRARequest("tinyllama-gqa-test", + lora_req = LoRARequest("llama-gqa-test", 0, str(nemo_path), lora_ckpt_source="nemo") @@ -986,21 +718,14 @@ def test_gqa_nemo_lora(tmp_path, cuda_graph_config): sampling_params, lora_request=[lora_req]) - # For the above deterministic dummy LoRA checkpoint, - # with temperature=0.0, - # the expected output text should always be the same. - assert lora_outputs[0].outputs[0].text == expected_lora_text_output, \ - f"Expected output text: {expected_lora_text_output}, " \ - f"got: {lora_outputs[0].outputs[0].text}" assert len(lora_outputs) == 1 # Generate without LoRA. # The LoRA weights are tuned/large enough that # they differ from a no-LoRA run. base_outputs = llm.generate(test_prompts, sampling_params) - assert base_outputs[0].outputs[0].text != expected_lora_text_output, \ - f"No-LoRA output should differ from expected output text: {expected_lora_text_output}, " \ - f"got: {base_outputs[0].outputs[0].text}" + assert lora_outputs[0].outputs[0].text != base_outputs[0].outputs[0].text, \ + "No-LoRA output should differ from LoRA output — adapter may not be applied" finally: llm.shutdown() diff --git a/tests/unittest/llmapi/test_memory_profiling.py b/tests/unittest/llmapi/test_memory_profiling.py index 3b0300634848..4f440bfd5871 100644 --- a/tests/unittest/llmapi/test_memory_profiling.py +++ b/tests/unittest/llmapi/test_memory_profiling.py @@ -83,7 +83,7 @@ def test_pyexecutor_and_kvcache_share_execution_stream(): Both components must use the same stream for proper synchronization. """ # Use a simple model for testing - MODEL = "llama-3.2-models/Llama-3.2-1B-Instruct" + MODEL = "llama-3.1-model/Llama-3.1-8B-Instruct" MODEL_PATH = get_model_path(MODEL) kv_cache_config = KvCacheConfig(enable_block_reuse=False, diff --git a/tests/unittest/test_pip_install.py b/tests/unittest/test_pip_install.py index e15dfcbb3eee..12d009dc323e 100644 --- a/tests/unittest/test_pip_install.py +++ b/tests/unittest/test_pip_install.py @@ -215,7 +215,9 @@ def create_link_for_models(): print(f"ERROR: Models root {models_root} does not exist") exit(1) src_dst_dict = { - # TinyLlama-1.1B-Chat-v1.0 + # quickstart_example.py loads "TinyLlama/TinyLlama-1.1B-Chat-v1.0" by + # HuggingFace id; symlink it to the local copy so pip install sanity + # checks don't attempt a network download on air-gapped nodes. f"{models_root}/llama-models-v2/TinyLlama-1.1B-Chat-v1.0": f"{os.getcwd()}/TinyLlama/TinyLlama-1.1B-Chat-v1.0", }