diff --git a/.github/workflows/nightly-benchmark.yml b/.github/workflows/nightly-benchmark.yml index a7f12a0b68..e6012ede2c 100644 --- a/.github/workflows/nightly-benchmark.yml +++ b/.github/workflows/nightly-benchmark.yml @@ -121,6 +121,9 @@ jobs: - { id: Qwen/Qwen2.5-7B-Instruct, slug: Qwen-Qwen2.5-7B-Instruct, test_class: TestNightlyQwen7bSingle } - { id: Qwen/Qwen3-30B-A3B, slug: Qwen-Qwen3-30B-A3B, test_class: TestNightlyQwen30bSingle } - { id: openai/gpt-oss-20b, slug: openai-gpt-oss-20b, test_class: TestNightlyGptOss20bSingle } + - { id: meta-llama/Llama-4-Scout-17B-16E-Instruct, slug: meta-llama-Llama-4-Scout-17B-16E-Instruct, test_class: TestNightlyLlama4ScoutSingle } + - { id: meta-llama/Llama-3.3-70B-Instruct, slug: meta-llama-Llama-3.3-70B-Instruct, test_class: TestNightlyLlama70bSingle } + - { id: RedHatAI/Llama-3.3-70B-Instruct-FP8-dynamic, slug: RedHatAI-Llama-3.3-70B-Instruct-FP8-dynamic, test_class: TestNightlyLlama70bFp8Single } variant: - { id: sglang, runtime: sglang, grpc_only: "false", setup_vllm: false, setup_trtllm: false } - { id: vllm, runtime: vllm, grpc_only: "false", setup_vllm: true, setup_trtllm: false } diff --git a/e2e_test/benchmarks/test_nightly_perf.py b/e2e_test/benchmarks/test_nightly_perf.py index ef439eebb3..51b3d290a0 100644 --- a/e2e_test/benchmarks/test_nightly_perf.py +++ b/e2e_test/benchmarks/test_nightly_perf.py @@ -109,6 +109,27 @@ def _run_nightly(setup_backend, genai_bench_runner, model_id, worker_count=1, ** ["http", "grpc"], {}, ), + ( + "meta-llama/Llama-4-Scout-17B-16E-Instruct", + "Llama4Scout", + 1, + ["http", "grpc"], + {}, + ), + ( + "meta-llama/Llama-3.3-70B-Instruct", + "Llama70b", + 1, + ["http", "grpc"], + {}, + ), + ( + "RedHatAI/Llama-3.3-70B-Instruct-FP8-dynamic", + "Llama70bFp8", + 1, + ["http", "grpc"], + {}, + ), ] diff --git a/e2e_test/infra/model_specs.py b/e2e_test/infra/model_specs.py index 31ec82591b..64ee4ee1f5 100644 --- a/e2e_test/infra/model_specs.py +++ b/e2e_test/infra/model_specs.py @@ -102,14 +102,6 @@ def _resolve_model_path(hf_path: str) -> str: "tp": 1, "features": ["chat", "streaming", "multimodal"], }, - # Llama-4-Scout (17B with 16 experts) - Multimodal tests - "meta-llama/Llama-4-Scout-17B-16E-Instruct": { - "model": _resolve_model_path("meta-llama/Llama-4-Scout-17B-16E-Instruct"), - "tp": 4, - "features": ["chat", "streaming", "multimodal", "moe"], - "vllm_args": ["--max-model-len", "196608"], - "startup_timeout": 1200, - }, # Llama-4-Maverick (17B with 128 experts, FP8) - Nightly benchmarks "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8": { "model": _resolve_model_path("meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8"), @@ -126,6 +118,52 @@ def _resolve_model_path(hf_path: str) -> str: "--max-model-len=163840", # 160K context length (vLLM) "--attention-backend=FLASHINFER", # FLASHINFER attention backend ], + "startup_timeout": 1200, # Large MoE model may need extra download/load time + }, + # Llama-4-Scout (17B with 16 experts) - Nightly benchmarks and Multimodal tests + "meta-llama/Llama-4-Scout-17B-16E-Instruct": { + "model": _resolve_model_path("meta-llama/Llama-4-Scout-17B-16E-Instruct"), + "tp": 4, + "features": ["chat", "streaming", "function_calling", "multimodal", "moe"], + "worker_args": [ + "--context-length=196608", + "--attention-backend=fa3", + "--cuda-graph-max-bs=256", + "--max-running-requests=300", + "--mem-fraction-static=0.85", + ], + "vllm_args": [ + "--max-model-len=196608", + ], + "startup_timeout": 1200, # Large MoE model may need extra download/load time + }, + # Llama-3.3-70B - Nightly benchmarks + "meta-llama/Llama-3.3-70B-Instruct": { + "model": _resolve_model_path("meta-llama/Llama-3.3-70B-Instruct"), + "tp": 4, + "features": ["chat", "streaming", "function_calling"], + "worker_args": [ + "--mem-fraction-static=0.9", + ], + "vllm_args": [ + "--max-model-len=131072", + "--gpu-memory-utilization=0.9", + "--enable-chunked-prefill", + ], + }, + # Llama-3.3-70B FP8 - Nightly benchmarks + "RedHatAI/Llama-3.3-70B-Instruct-FP8-dynamic": { + "model": _resolve_model_path("RedHatAI/Llama-3.3-70B-Instruct-FP8-dynamic"), + "tp": 4, + "features": ["chat", "streaming", "function_calling"], + "worker_args": [ + "--mem-fraction-static=0.9", + ], + "vllm_args": [ + "--max-model-len=131072", + "--gpu-memory-utilization=0.9", + "--enable-chunked-prefill", + ], }, }