Skip to content
1 change: 1 addition & 0 deletions .github/workflows/nightly-benchmark.yml
Original file line number Diff line number Diff line change
Expand Up @@ -350,6 +350,7 @@ jobs:
- { id: meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8, slug: meta-llama-Llama-4-Maverick-17B-128E-Instruct-FP8, test_class: TestNightlyLlama4MaverickSingle }
# Pausing MiniMax M2 nightly benchmark
# - { id: minimaxai/minimax-m2, slug: minimaxai-minimax-m2, test_class: TestNightlyMinimaxM2Single }
- { id: zai-org/GLM-4.6, slug: zai-org-GLM-4.6, test_class: TestNightlyGlm46Single }
variant:
- { id: sglang, runtime: sglang, grpc_only: "false", setup_vllm: false, setup_trtllm: false, extra_deps: "genai-bench" }
- { id: vllm, runtime: vllm, grpc_only: "false", setup_vllm: true, setup_trtllm: false, extra_deps: "genai-bench" }
Expand Down
11 changes: 9 additions & 2 deletions e2e_test/benchmarks/test_nightly_perf.py
Original file line number Diff line number Diff line change
Expand Up @@ -102,6 +102,7 @@ def _run_nightly(setup_backend, genai_bench_runner, model_id, worker_count=1, **
("Qwen/Qwen3-30B-A3B", "Qwen30b", 4, ["http", "grpc"], {}),
("openai/gpt-oss-20b", "GptOss20b", 1, ["http", "grpc"], {}),
("minimaxai/minimax-m2", "MinimaxM2", 1, ["http", "grpc"], {}),
("zai-org/GLM-4.6", "Glm46", 1, ["http", "grpc"], {}),
(
"meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8",
"Llama4Maverick",
Expand Down Expand Up @@ -132,6 +133,8 @@ def _run_nightly(setup_backend, genai_bench_runner, model_id, worker_count=1, **
),
]

_SINGLE_ONLY_NIGHTLY_MODELS = {"zai-org/GLM-4.6"}


# ---------------------------------------------------------------------------
# Dynamic test class generation
Expand Down Expand Up @@ -163,12 +166,16 @@ def test_nightly_perf(self, setup_backend, genai_bench_runner):


for _model_id, _name, _multi_workers, _backends, _extra in _NIGHTLY_MODELS:
for _suffix, _count in [("Single", 1), ("Multi", _multi_workers)]:
variants = [("Single", 1)]
if _model_id not in _SINGLE_ONLY_NIGHTLY_MODELS:
variants.append(("Multi", _multi_workers))

for _suffix, _count in variants:
_cls_name = f"TestNightly{_name}{_suffix}"
_cls = _make_test_class(_model_id, _count, _backends, _extra)
_cls.__name__ = _cls_name
_cls.__qualname__ = _cls_name
globals()[_cls_name] = _cls

# Clean up loop variables from module namespace
del _model_id, _name, _multi_workers, _backends, _extra, _suffix, _count, _cls_name, _cls
del _model_id, _name, _multi_workers, _backends, _extra, _suffix, _count, _cls_name, _cls, variants
8 changes: 8 additions & 0 deletions e2e_test/infra/model_specs.py
Original file line number Diff line number Diff line change
Expand Up @@ -110,6 +110,14 @@ def _resolve_model_path(hf_path: str) -> str:
"worker_args": ["--trust-remote-code"],
"vllm_args": ["--trust-remote-code"],
},
# GLM-4.6 - nightly benchmarks
"zai-org/GLM-4.6": {
"model": _resolve_model_path("zai-org/GLM-4.6"),
"tp": 4,

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

This model is 665GB in BF16. It won't fit in 8xH100.

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Confirmed with @slin1237 and it looks like it needs multi node support, which is not supported by the nightly benchmarking pipeline.

Closing this PR

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Thanks for the review @CatherineSue

"features": ["chat", "streaming", "function_calling", "reasoning"],
"worker_args": ["--trust-remote-code"],
"vllm_args": ["--trust-remote-code"],

@CatherineSue CatherineSue Apr 1, 2026 •

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

We don't need this parameter for this one. No .py in the model files.

},
Comment thread
nishanthp marked this conversation as resolved.
# Vision-language model for multimodal benchmarks (MMMU)
"Qwen/Qwen3-VL-8B-Instruct": {
"model": _resolve_model_path("Qwen/Qwen3-VL-8B-Instruct"),
Expand Down
Loading