Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion python/sglang/srt/environ.py
Original file line number Diff line number Diff line change
Expand Up @@ -1359,7 +1359,7 @@ class Envs:
# capacity).
SGLANG_DISABLE_DRAFT_EXTEND_CUDA_GRAPH = EnvBool(False)
# Use the split-KV (flash-decode) kernel for EAGLE target-verify on the
# Triton backend (ROCm). Only active at speculative topk == 1; falls back to
# Triton backend (ROCm and CUDA). Only active at speculative topk == 1; falls back to
# extend_attention_fwd for unsupported cases or when set false (e.g. for
# debugging). Correctness is unaffected; this only changes performance.
SGLANG_ENABLE_SPLITKV_VERIFY = EnvBool(True)
Expand Down
12 changes: 7 additions & 5 deletions python/sglang/srt/layers/attention/triton_backend.py
Original file line number Diff line number Diff line change
Expand Up @@ -222,12 +222,14 @@ def __init__(
self.target_verify_num_tokens_per_req = model_runner.decode_num_tokens_per_req()
self.speculative_num_steps = get_spec().speculative_num_steps
self.topk = get_spec().speculative_eagle_topk or 0
# Split-KV verify is bit-equivalent only for a pure-causal chain (topk==1)
# and is gfx95-only; else fall back to extend_attention_fwd.
# Split-KV verify is bit-equivalent only for a pure-causal chain (topk==1);
# else fall back to extend_attention_fwd. The kernel is NV-safe (HIP-only
# launch kwargs are gated inside verify_splitkv.py), and on CUDA the
# fallback extend kernel loops the whole prefix serially per (seq, head),
# so long-context verify degrades linearly with prefix length. Opt out
# with SGLANG_ENABLE_SPLITKV_VERIFY=0.
self.use_verify_splitkv = (
is_gfx95_supported()
and envs.SGLANG_ENABLE_SPLITKV_VERIFY.get()
and self.topk == 1
envs.SGLANG_ENABLE_SPLITKV_VERIFY.get() and self.topk == 1
)
self.use_mla = model_runner.model_config.attention_arch == AttentionArch.MLA
# The grouped-head verify kernel is tuned for Kimi-K3 MLA and Qwen3.5
Expand Down
Loading