From b56dc8c71e00122547d773ff0d9bef6ac011e2dd Mon Sep 17 00:00:00 2001 From: ZhanruiSunCh <184402041+ZhanruiSunCh@users.noreply.github.com> Date: Fri, 10 Apr 2026 12:20:46 +0800 Subject: [PATCH] [None][infra] Waive 52 failed cases for release/1.2.1 in post-merge 3 Bug(s): 6065451 Requested by: @mzweilz Signed-off-by: ZhanruiSunCh <184402041+ZhanruiSunCh@users.noreply.github.com> --- tests/integration/test_lists/waives.txt | 52 +++++++++++++++++++++++++ 1 file changed, 52 insertions(+) diff --git a/tests/integration/test_lists/waives.txt b/tests/integration/test_lists/waives.txt index 7a318ad331c3..f0329c90cc61 100644 --- a/tests/integration/test_lists/waives.txt +++ b/tests/integration/test_lists/waives.txt @@ -352,3 +352,55 @@ accuracy/test_llm_api_pytorch.py::TestQwen3_235B_A22B::test_nvfp4_4gpus[latency_ unittest/_torch/modeling -k "modeling_siglip" SKIP (https://nvbugs/5941242) stress_test/stress_test.py::test_run_stress_test[llama-v3-8b-instruct-hf_tp1-stress_time_300s_timeout_450s-GUARANTEED_NO_EVICT-pytorch-stress-test] SKIP (https://nvbugs/5955927) stress_test/stress_test.py::test_run_stress_test[llama-v3-8b-instruct-hf_tp1-stress_time_300s_timeout_450s-MAX_UTILIZATION-pytorch-stress-test] SKIP (https://nvbugs/5955927) +examples/test_llama.py::test_llm_llama_v3_1_1node_single_gpu[llama-3.2-1b-disable_fp8] SKIP (https://nvbugs/6065451) +examples/test_llama.py::test_llm_llama_1gpu[llama-3.1-8b-instruct-hf-fp8-enable_fp8-float16-summarization-nb:1] SKIP (https://nvbugs/6065451) +examples/test_llama.py::test_llm_llama_v1_1gpu_kv_cache_reuse_with_prompt_table[llama-7b] SKIP (https://nvbugs/6065451) +perf/test_perf.py::test_perf[llama_v3.1_8b_instruct-bench-float16-input_output_len:128,128-reqs:8192] SKIP (https://nvbugs/6065451) +test_e2e.py::test_trtllm_bench_request_rate_and_concurrency[enable_concurrency-enable_request_rate] SKIP (https://nvbugs/6065451) +test_e2e.py::test_trtllm_bench_request_rate_and_concurrency[enable_concurrency-] SKIP (https://nvbugs/6065451) +cpp/test_e2e.py::test_model[-bart-90] SKIP (https://nvbugs/6065451) +cpp/test_e2e.py::test_benchmarks[t5-90] SKIP (https://nvbugs/6065451) +cpp/test_e2e.py::test_model[-enc_dec_language_adapter-90] SKIP (https://nvbugs/6065451) +cpp/test_e2e.py::test_model[-t5-90] SKIP (https://nvbugs/6065451) +cpp/test_e2e.py::test_model[fp8-llama-90] SKIP (https://nvbugs/6065451) +triton_server/test_triton.py::test_gpt[gpt] SKIP (https://nvbugs/6065451) +triton_server/test_triton.py::test_gptj[gptj] SKIP (https://nvbugs/6065451) +triton_server/test_triton.py::test_mistral[mistral] SKIP (https://nvbugs/6065451) +triton_server/test_triton.py::test_mistral_ib_streaming[mistral-ib-streaming] SKIP (https://nvbugs/6065451) +triton_server/test_triton.py::test_whisper[whisper] SKIP (https://nvbugs/6065451) +triton_server/test_triton_llm.py::test_benchmark_core_model[llama_v2_7b-False-1---False-True-False-0-128-disableDecoupleMode-inflight_fused_batching-disableTrtOverlap--max_utilization-4096--1-1-1-False] SKIP (https://nvbugs/6065451) +triton_server/test_triton_llm.py::test_eagle_vicuna_7b_ifb[False-1-eagle--False-True-False-0-128-enableDecoupleMode-inflight_fused_batching-disableTrtOverlap--max_utilization---1-1-1-False-ensemble] SKIP (https://nvbugs/6065451) +triton_server/test_triton_llm.py::test_gpt_350m_python_backend[accuracy] SKIP (https://nvbugs/6065451) +triton_server/test_triton_llm.py::test_gpt_speculative_decoding_bls[True-False-1---False-True-True-0-128-disableDecoupleMode-inflight_fused_batching-disableTrtOverlap-0.2-guaranteed_no_evict---1-1-1-False-ensemble] SKIP (https://nvbugs/6065451) +triton_server/test_triton_llm.py::test_llama_v2_7b_ifb[batched_inputs-False-1---False-True-False-0-128-enableDecoupleMode-inflight_fused_batching-enableTrtOverlap--max_utilization---1-1-1-True-tensorrt_llm_bls] SKIP (https://nvbugs/6065451) +triton_server/test_triton_llm.py::test_mistral_v1_7b_python_backend[accuracy] SKIP (https://nvbugs/6065451) +triton_server/test_triton_llm.py::test_t5_small_enc_dec_ifb[test_basic-False-1-top_k_top_p--False-True-False-0-128-disableDecoupleMode-inflight_fused_batching-disableTrtOverlap---guaranteed_no_evict--4096-1-1-1-False-tensorrt_llm_bls] SKIP (https://nvbugs/6065451) +triton_server/test_triton_llm.py::test_tiny_llama_1b_guided_decoding[xgrammar-tensorrtllm-True-1---False-True-False-0-128-disableDecoupleMode-inflight_fused_batching-disableTrtOverlap--guaranteed_no_evict---1-1-1-False-ensemble-accuracy] SKIP (https://nvbugs/6065451) +triton_server/test_triton_llm.py::test_tiny_llama_ifb_token_counts[python-both-False-1---False-True-False-0-128-enableDecoupleMode-inflight_fused_batching-disableTrtOverlap--guaranteed_no_evict---1-1-1-False-ensemble] SKIP (https://nvbugs/6065451) +triton_server/test_triton_llm.py::test_tiny_llama_ifb_token_counts[tensorrtllm-both-False-1---False-True-False-0-128-disableDecoupleMode-inflight_fused_batching-disableTrtOverlap--guaranteed_no_evict---1-1-1-False-tensorrt_llm_bls] SKIP (https://nvbugs/6065451) +triton_server/test_triton_llm.py::test_tiny_llama_ifb_token_counts[tensorrtllm-both-False-1---False-True-False-0-128-enableDecoupleMode-inflight_fused_batching-disableTrtOverlap--guaranteed_no_evict---1-1-1-False-tensorrt_llm_bls] SKIP (https://nvbugs/6065451) +triton_server/test_triton.py::test_llama[llama] SKIP (https://nvbugs/6065451) +triton_server/test_triton_llm.py::test_benchmark_core_model[gptj_6b-False-1---False-True-False-0-128-disableDecoupleMode-inflight_fused_batching-disableTrtOverlap--max_utilization-4096--1-1-1-False] SKIP (https://nvbugs/6065451) +triton_server/test_triton_llm.py::test_gpt_350m_ifb[test_basic-False-1---False-True-False-0-128-enableDecoupleMode-inflight_fused_batching-disableTrtOverlap--max_utilization---1-1-1-True-tensorrt_llm_bls] SKIP (https://nvbugs/6065451) +triton_server/test_triton_llm.py::test_gpt_350m_python_backend[e2e] SKIP (https://nvbugs/6065451) +triton_server/test_triton_llm.py::test_gpt_disaggregated_serving_bls[test_basic-False-1-top_k_top_p--False-True-True-0-128-enableDecoupleMode-inflight_fused_batching-disableTrtOverlap-0.2-max_utilization---1-1-1-True-tensorrt_llm_bls] SKIP (https://nvbugs/6065451) +triton_server/test_triton_llm.py::test_gpt_speculative_decoding_bls[False-False-1---False-True-True-0-128-disableDecoupleMode-inflight_fused_batching-disableTrtOverlap-0.2-guaranteed_no_evict---1-1-1-False-ensemble] SKIP (https://nvbugs/6065451) +triton_server/test_triton_llm.py::test_gpt_speculative_decoding_bls[True-False-1---False-True-True-0-128-disableDecoupleMode-inflight_fused_batching-disableTrtOverlap-0.2-max_utilization---1-1-1-False-ensemble] SKIP (https://nvbugs/6065451) +triton_server/test_triton_llm.py::test_llama_v2_7b_ifb[batched_inputs-False-1---False-True-False-0-128-enableDecoupleMode-inflight_fused_batching-disableTrtOverlap--max_utilization---1-1-1-True-tensorrt_llm_bls] SKIP (https://nvbugs/6065451) +triton_server/test_triton_llm.py::test_llama_v2_7b_ifb[test_stop_words-False-1---False-True-False-0-128-enableDecoupleMode-inflight_fused_batching-disableTrtOverlap--max_utilization---1-1-1-True-tensorrt_llm_bls] SKIP (https://nvbugs/6065451) +triton_server/test_triton_llm.py::test_medusa_vicuna_7b_ifb[False-1-medusa--False-True-False-0-128-enableDecoupleMode-inflight_fused_batching-disableTrtOverlap--max_utilization---1-1-1-False-ensemble] SKIP (https://nvbugs/6065451) +triton_server/test_triton_llm.py::test_mistral_v1_7b_ifb[False-1---False-True-False-0-128-enableDecoupleMode-inflight_fused_batching-disableTrtOverlap--max_utilization-4096--1-1-1-False-ensemble] SKIP (https://nvbugs/6065451) +triton_server/test_triton_llm.py::test_mistral_v1_7b_python_backend[e2e] SKIP (https://nvbugs/6065451) +triton_server/test_triton_llm.py::test_tiny_llama_1b_guided_decoding[xgrammar-python-True-1---False-True-False-0-128-disableDecoupleMode-inflight_fused_batching-disableTrtOverlap--guaranteed_no_evict---1-1-1-False-ensemble-accuracy] SKIP (https://nvbugs/6065451) +triton_server/test_triton_llm.py::test_tiny_llama_ifb_token_counts[python-both-False-1---False-True-False-0-128-disableDecoupleMode-inflight_fused_batching-disableTrtOverlap--guaranteed_no_evict---1-1-1-False-tensorrt_llm_bls] SKIP (https://nvbugs/6065451) +triton_server/test_triton_llm.py::test_tiny_llama_ifb_token_counts[python-both-False-1---False-True-False-0-128-enableDecoupleMode-inflight_fused_batching-disableTrtOverlap--guaranteed_no_evict---1-1-1-False-tensorrt_llm_bls] SKIP (https://nvbugs/6065451) +triton_server/test_triton_llm.py::test_tiny_llama_ifb_token_counts[tensorrtllm-both-False-1---False-True-False-0-128-enableDecoupleMode-inflight_fused_batching-disableTrtOverlap--guaranteed_no_evict---1-1-1-False-ensemble] SKIP (https://nvbugs/6065451) +triton_server/test_triton_llm.py::test_whisper_large_v3_ifb[True-1-top_k_top_p--False-True-False-0-128-disableDecoupleMode-inflight_fused_batching-disableTrtOverlap-0.2-0.5-guaranteed_no_evict--24000-1-1-1-False-ensemble] SKIP (https://nvbugs/6065451) +triton_server/test_triton.py::test_opt[opt] SKIP (https://nvbugs/6065451) +cpp/test_e2e.py::test_model[-gpt_executor-80] SKIP (https://nvbugs/6065451) +cpp/test_e2e.py::test_model[-gpt_tests-80] SKIP (https://nvbugs/6065451) +triton_server/test_triton.py::test_medusa[medusa] SKIP (https://nvbugs/6065451) +cpp/test_e2e.py::test_model[-eagle-86] SKIP (https://nvbugs/6065451) +cpp/test_e2e.py::test_model[-medusa-86] SKIP (https://nvbugs/6065451) +unittest/bindings/test_executor_bindings.py::test_embedding_bias SKIP (https://nvbugs/6065451) +unittest/bindings/test_executor_bindings.py::test_executor_from_memory SKIP (https://nvbugs/6065451)