Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion tests/e2e/pull_request/four_card/test_qwen3_5.py
Original file line number Diff line number Diff line change
Expand Up @@ -43,7 +43,7 @@ def test_qwen3_5_27b_distributed_mp_tp4():
del vllm_model


def test_qwen3_5_35b_distributed_mp_tp4_full_decode_only_mtp3():
def test_qwen3_5_35b_dp2_tp2_ep_sp_full_decode_only_mtp3():
example_prompts = [
"2 + 2 =",
"The president of the United States is",
Expand All @@ -57,6 +57,7 @@ def test_qwen3_5_35b_distributed_mp_tp4_full_decode_only_mtp3():
data_parallel_size=2,
tensor_parallel_size=2,
enable_expert_parallel=True,
all2all_backend="allgather_reducescatter",
max_model_len=4096,
gpu_memory_utilization=0.90,
distributed_executor_backend="mp",
Expand Down
19 changes: 19 additions & 0 deletions tests/ut/ops/test_register_custom_ops.py
Original file line number Diff line number Diff line change
Expand Up @@ -62,6 +62,25 @@ def test_sp_ep_reduce_scatter_pads_local_chunks(monkeypatch):
assert result.shape == (3, 4)


def test_sp_ep_reduce_scatter_draft_model_keeps_ep_layout(monkeypatch):
_patch_sp_ep_context(monkeypatch)
custom_ops._EXTRA_CTX.is_draft_model = True

def unexpected_tp_all_reduce(_x):
raise AssertionError("EP/SP finalize must not use TP AllReduce")

monkeypatch.setattr(
custom_ops,
"tensor_model_parallel_all_reduce",
unexpected_tp_all_reduce,
raising=False,
)

result = custom_ops._maybe_pad_and_reduce_impl(torch.arange(32).view(8, 4))

assert result.shape == (3, 4)


def test_sp_ep_reduce_scatter_unpads_local_chunk(monkeypatch):
_patch_sp_ep_context(monkeypatch)
monkeypatch.setattr(custom_ops, "get_ep_group", _EpGroupRank0)
Expand Down
4 changes: 0 additions & 4 deletions vllm_ascend/ops/register_custom_ops.py
Original file line number Diff line number Diff line change
Expand Up @@ -12,7 +12,6 @@
from vllm_ascend.ascend_forward_context import _EXTRA_CTX, MoECommType
from vllm_ascend.ops.rotary_embedding import rope_forward_oot
from vllm_ascend.ops.triton.muls_add import muls_add_triton
from vllm_ascend.utils import is_vl_model


def _get_ep_local_sizes(dp_metadata, ep_group) -> list[int] | None:
Expand Down Expand Up @@ -90,9 +89,6 @@ def _maybe_pad_and_reduce_impl(x: torch.Tensor) -> torch.Tensor:
"""EP communication only: pad according to the DP token distribution, then EP reduce_scatter."""
forward_context = get_forward_context()

if _EXTRA_CTX.is_draft_model and is_vl_model():
return tensor_model_parallel_all_reduce(x)

dp_metadata = forward_context.dp_metadata
if dp_metadata is None:
return get_ep_group().reduce_scatter(x, 0)
Expand Down
Loading