diff --git a/python/sglang/srt/environ.py b/python/sglang/srt/environ.py index e874228d27bc..ec3a56041eeb 100644 --- a/python/sglang/srt/environ.py +++ b/python/sglang/srt/environ.py @@ -1359,7 +1359,7 @@ class Envs: # capacity). SGLANG_DISABLE_DRAFT_EXTEND_CUDA_GRAPH = EnvBool(False) # Use the split-KV (flash-decode) kernel for EAGLE target-verify on the - # Triton backend (ROCm). Only active at speculative topk == 1; falls back to + # Triton backend (ROCm and CUDA). Only active at speculative topk == 1; falls back to # extend_attention_fwd for unsupported cases or when set false (e.g. for # debugging). Correctness is unaffected; this only changes performance. SGLANG_ENABLE_SPLITKV_VERIFY = EnvBool(True) diff --git a/python/sglang/srt/layers/attention/triton_backend.py b/python/sglang/srt/layers/attention/triton_backend.py index 914def77f6b9..8ffa62310281 100644 --- a/python/sglang/srt/layers/attention/triton_backend.py +++ b/python/sglang/srt/layers/attention/triton_backend.py @@ -222,12 +222,14 @@ def __init__( self.target_verify_num_tokens_per_req = model_runner.decode_num_tokens_per_req() self.speculative_num_steps = get_spec().speculative_num_steps self.topk = get_spec().speculative_eagle_topk or 0 - # Split-KV verify is bit-equivalent only for a pure-causal chain (topk==1) - # and is gfx95-only; else fall back to extend_attention_fwd. + # Split-KV verify is bit-equivalent only for a pure-causal chain (topk==1); + # else fall back to extend_attention_fwd. The kernel is NV-safe (HIP-only + # launch kwargs are gated inside verify_splitkv.py), and on CUDA the + # fallback extend kernel loops the whole prefix serially per (seq, head), + # so long-context verify degrades linearly with prefix length. Opt out + # with SGLANG_ENABLE_SPLITKV_VERIFY=0. self.use_verify_splitkv = ( - is_gfx95_supported() - and envs.SGLANG_ENABLE_SPLITKV_VERIFY.get() - and self.topk == 1 + envs.SGLANG_ENABLE_SPLITKV_VERIFY.get() and self.topk == 1 ) self.use_mla = model_runner.model_config.attention_arch == AttentionArch.MLA # The grouped-head verify kernel is tuned for Kimi-K3 MLA and Qwen3.5