Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -309,8 +309,6 @@ def __init__(self, runner: ModelRunner):
# that wide. Head split mirrors MiniMaxM3 sparse attention's.
self._idx_group_size = 1
if self.index_cache_enabled:
from sglang.srt.runtime_context import get_parallel

_num_idx_heads = max(
sparse_cfg["sparse_num_index_heads"] // get_parallel().attn_tp_size, 1
)
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,27 @@
import unittest

from sglang.test.ci.ci_register import register_cpu_ci
from sglang.test.test_utils import CustomTestCase

register_cpu_ci(est_time=5, suite="base-a-test-cpu")


class TestMiniMaxSparseBackendScoping(CustomTestCase):
"""MiniMaxSparseAttnBackend.__init__ calls the module-level get_parallel() on
the MSA path. A function-level import of get_parallel further down the same
method made the name local to all of __init__, so a GPU start that took the
MSA path raised UnboundLocalError. No CPU runner builds this backend, so the
scoping is checked directly."""

def test_get_parallel_is_not_a_local_of_init(self):
from sglang.srt.layers.attention.minimax_sparse_backend import (
MiniMaxSparseAttnBackend,
)

self.assertNotIn(
"get_parallel", MiniMaxSparseAttnBackend.__init__.__code__.co_varnames
)


if __name__ == "__main__":
unittest.main()
Loading