From 6a230557358a2958cd0608c0c7c323f58fddc7c8 Mon Sep 17 00:00:00 2001 From: Meenakshi Venkataraman Date: Mon, 21 Sep 2026 21:22:53 +0000 Subject: [PATCH] [Bugfix] Exclude SM107 (Rubin) from support_deep_gemm() is_device_capability_family(100) matches any 10.x capability, so it silently admits SM107 (Rubin) into the DeepGEMM/CuTe-gated code paths alongside SM100/SM103 (Blackwell). The vendored DeepGEMM/CuTe kernels are not built for native sm_107, so this causes a cuModuleLoadData -> CUDA_ERROR_ASSERT at kernel load time -- observed via the MHC TileLang path (mhc_pre_big_fuse_with_norm_tilelang) on SM107 hardware, but the same gate also feeds every other DeepGEMM call site behind is_deep_gemm_supported(). Replace the family-wide check with explicit SM100/SM103 entries, matching how Hopper (90) and the 120 family are already enumerated individually in this function. When DeepGEMM ships kernels built for native sm_107, add it back the same way. Co-Authored-By: Claude Sonnet 5 Signed-off-by: Meenakshi Venkataraman --- vllm/platforms/cuda.py | 11 +++++++++-- 1 file changed, 9 insertions(+), 2 deletions(-) diff --git a/vllm/platforms/cuda.py b/vllm/platforms/cuda.py index a46b607c8e47..308b3980a046 100644 --- a/vllm/platforms/cuda.py +++ b/vllm/platforms/cuda.py @@ -717,10 +717,17 @@ def support_static_graph_mode(cls) -> bool: @classmethod def support_deep_gemm(cls) -> bool: - """Currently, only Hopper and Blackwell GPUs are supported.""" + """Currently, only Hopper and Blackwell GPUs are supported. + + Within the SM100 family, only SM100 and SM103 are supported: the + vendored DeepGEMM/CuTe kernels are not built for native sm_107, so + `is_device_capability_family(100)` (which matches any 10.x) would + wrongly admit it and crash with CUDA_ERROR_ASSERT at kernel load. + """ return ( cls.is_device_capability(90) - or cls.is_device_capability_family(100) + or cls.is_device_capability(100) + or cls.is_device_capability(103) or cls.is_device_capability_family(120) )