Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
16 changes: 14 additions & 2 deletions .buildkite/amd/test-amd-merge.yml
Original file line number Diff line number Diff line change
Expand Up @@ -9,22 +9,34 @@ steps:
steps:
- label: "Simple · Model Executor Test · Shard %N/%t"
agent_pool: mi300_1
parallelism: 2
parallelism: 3
depends_on: amd-build
mirror_hardwares: [amdproduction]
grade: Blocking
timeout_in_minutes: 90
commands:
- export VLLM_ROCM_USE_AITER=0
# ignore test_teacache_extractors.py because it use rocm gemm kernel from vLLM
# that is not supported on CPU
- "pytest -sv tests/model_executor -m 'core_model and cpu' --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB"
- "pytest -sv tests/model_executor -m 'core_model and cpu and not omni' --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB"

- label: "Simple · Omni Processor Test"
agent_pool: mi300_1
depends_on: amd-build
mirror_hardwares: [amdproduction]
grade: Blocking
timeout_in_minutes: 45
commands:
- export VLLM_ROCM_USE_AITER=0
- "pytest -sv tests/model_executor/models/test_omni_processing.py -m 'core_model and cpu'"

- label: "Simple · Diffusion Test · Shard %N/%t"
agent_pool: mi300_1
parallelism: 4
depends_on: amd-build
mirror_hardwares: [amdproduction]
grade: Blocking
timeout_in_minutes: 30
commands:
- export VLLM_ROCM_USE_AITER=0
# ignore test_teacache_extractors.py because it use rocm gemm kernel from vLLM
Expand Down
37 changes: 35 additions & 2 deletions .buildkite/amd/test-amd-ready.yml
Original file line number Diff line number Diff line change
Expand Up @@ -9,21 +9,33 @@ steps:
steps:
- label: "Simple · Model Executor Test · Shard %N/%t"
agent_pool: mi300_1
parallelism: 2
parallelism: 3
depends_on: amd-build
mirror_hardwares: [amdproduction]
grade: Blocking
timeout_in_minutes: 90
commands:
- export VLLM_ROCM_USE_AITER=0
# ignore test_teacache_extractors.py because it use rocm gemm kernel from vLLM
# that is not supported on CPU
- "pytest -sv tests/model_executor -m 'core_model and cpu' --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB"
- "pytest -sv tests/model_executor -m 'core_model and cpu and not omni' --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB"

- label: "Simple · Omni Processor Test"
agent_pool: mi300_1
depends_on: amd-build
mirror_hardwares: [amdproduction]
grade: Blocking
timeout_in_minutes: 45
commands:
- export VLLM_ROCM_USE_AITER=0
- "pytest -sv tests/model_executor/models/test_omni_processing.py -m 'core_model and cpu'"

- label: "Simple · Diffusion Test"
agent_pool: mi300_1
depends_on: amd-build
mirror_hardwares: [amdproduction]
grade: Blocking
timeout_in_minutes: 60
commands:
- export VLLM_ROCM_USE_AITER=0
# ignore test_teacache_extractors.py because it use rocm gemm kernel from vLLM
Expand Down Expand Up @@ -69,6 +81,27 @@ steps:

- group: ":card_index_dividers: Diffusion Test"
steps:
# Keep real FlashAttention execution out of the CPU-marked Simple lane.
# AITER may need to build several kernels on a cold worker, so run one
# pytest worker with job-local build directories and a bounded timeout.
- label: "ROCm · FlashAttention/AITER GPU Coverage"
agent_pool: mi300_1
depends_on: amd-build
mirror_hardwares: [amdproduction]
grade: NonBlocking
timeout_in_minutes: 60
commands:
- export VLLM_ROCM_USE_AITER=1
- export AITER_JIT_DIR="/tmp/vllm-omni-aiter-$$BUILDKITE_JOB_ID"
- export TORCH_EXTENSIONS_DIR="/tmp/vllm-omni-torch-extensions-$$BUILDKITE_JOB_ID"
- mkdir -p "$$AITER_JIT_DIR" "$$TORCH_EXTENSIONS_DIR"
- >-
timeout --signal=TERM --kill-after=2m 55m
pytest -s -v -n 1
tests/diffusion/attention/
-m 'core_model and rocm and MI325 and cards_1'
--run-level "core_model"

- label: "Diffusion · Batch Test"
agent_pool: mi300_1
depends_on: amd-build
Expand Down
6 changes: 6 additions & 0 deletions .buildkite/cuda/test-ready.yml
Original file line number Diff line number Diff line change
Expand Up @@ -38,6 +38,12 @@ steps:
depends_on: upload-ready-pipeline
steps:

- label: "Diffusion · FlashAttention Test"
timeout_in_minutes: 20
commands:
- timeout 15m pytest -s -v tests/diffusion/attention/ -m 'core_model and cuda and L4 and cards_1' --run-level "core_model"
mirror_hardwares: l4_1

- label: "Diffusion · Offloader Test"
commands:
- timeout 40m pytest -s -v tests/diffusion/offloader -m "core_model and cuda" --run-level "core_model"
Expand Down
33 changes: 28 additions & 5 deletions tests/diffusion/attention/test_flash_attn.py
Original file line number Diff line number Diff line change
Expand Up @@ -16,13 +16,14 @@
import pytest
import torch

from tests.helpers.mark import hardware_test
from vllm_omni.diffusion.attention.backends.abstract import AttentionMetadata
from vllm_omni.diffusion.attention.backends.flash_attn import FlashAttentionImpl
from vllm_omni.diffusion.attention.backends.sdpa import SDPAImpl
from vllm_omni.diffusion.attention.backends.utils import fa # noqa: E402
from vllm_omni.platforms import current_omni_platform

pytestmark = [pytest.mark.core_model, pytest.mark.diffusion, pytest.mark.cpu]
pytestmark = [pytest.mark.core_model, pytest.mark.diffusion]

is_gpu = current_omni_platform.is_cuda_alike() or current_omni_platform.is_xpu()
HAS_FLASH_ATTN = fa.HAS_FLASH_ATTN
Expand All @@ -42,10 +43,12 @@
(9, "4", frozenset((2, 3, 4)), 4),
],
)
@pytest.mark.cpu
def test_versioned_flash_attention_selection(device_major, requested, supported, expected):
assert fa._choose_vllm_flash_attn_version(device_major, requested, supported) == expected


@pytest.mark.cpu
def test_versioned_flash_attention_selection_rejects_invalid_or_unavailable_version():
with pytest.raises(ValueError, match="must be 2, 3, or 4"):
fa._choose_vllm_flash_attn_version(9, "5", frozenset((2, 3, 4)))
Expand Down Expand Up @@ -92,7 +95,8 @@ def pad_tensor(tensor: torch.Tensor, target_seq_len: int, pad_value: float = 0.0
return torch.cat([tensor, padding], dim=1)


@pytest.mark.skipif(not is_gpu, reason="FlashAttention requires CUDA or XPU")
@hardware_test(res={"cuda": "L4", "rocm": "MI325"}, num_cards=1)
@pytest.mark.skipif(not is_gpu, reason="FlashAttention requires an accelerator")
def test_padding_equivalence():
"""
Case 1: Test that padded and unpadded inputs produce similar outputs.
Expand Down Expand Up @@ -185,7 +189,8 @@ def test_padding_equivalence():
print("✓ Case 1 PASSED: Padded and unpadded outputs are very close!")


@pytest.mark.skipif(not is_gpu, reason="FlashAttention requires CUDA or XPU")
@hardware_test(res={"cuda": "L4", "rocm": "MI325"}, num_cards=1)
@pytest.mark.skipif(not is_gpu, reason="FlashAttention requires an accelerator")
def test_fa_vs_sdpa():
"""
Case 2: Compare FlashAttention and SDPA backends with padding.
Expand Down Expand Up @@ -302,7 +307,8 @@ def test_fa_vs_sdpa():
print("✓ Case 2 PASSED: FA and SDPA outputs are very close!")


@pytest.mark.skipif(not is_gpu, reason="FlashAttention requires CUDA or XPU")
@hardware_test(res={"cuda": "L4", "rocm": "MI325"}, num_cards=1)
@pytest.mark.skipif(not is_gpu, reason="FlashAttention requires an accelerator")
def test_flash_attn_func_preferred_over_varlen():
"""Test flash_attn_func availability and basic forward call."""
if not HAS_FLASH_ATTN:
Expand Down Expand Up @@ -332,7 +338,8 @@ def test_flash_attn_func_preferred_over_varlen():
print("✓ flash_attn_func forward works correctly!")


@pytest.mark.skipif(not is_gpu, reason="FlashAttention requires CUDA or XPU")
@hardware_test(res={"cuda": "L4", "rocm": "MI325"}, num_cards=1)
@pytest.mark.skipif(not is_gpu, reason="FlashAttention requires an accelerator")
@pytest.mark.parametrize("k_len", [40, 256])
def test_cross_attn_key_padding_vs_sdpa(k_len):
"""
Expand Down Expand Up @@ -390,6 +397,7 @@ def test_cross_attn_key_padding_vs_sdpa(k_len):
print("✓ Case 3 PASSED: FA and SDPA cross-attention outputs are very close!")


@pytest.mark.cpu
def test_varlen_masked_routing_by_role(monkeypatch):
"""The unpad route is picked by role, not Q/K length equality; runs without a GPU."""
calls = []
Expand Down Expand Up @@ -424,6 +432,7 @@ def fake_varlen_func(q, k, v, **kwargs):
assert cu_seqlens_k.tolist() == [0, 2, 5]


@pytest.mark.cpu
def test_piecewise_flash_attn_uses_varlen_fallback(monkeypatch):
calls = []

Expand Down Expand Up @@ -454,6 +463,7 @@ def fake_varlen_func(q, k, v, **kwargs):
assert calls[0]["softmax_scale"] == 0.5


@pytest.mark.cpu
def test_packed_varlen_metadata_bypasses_mask_unpadding(monkeypatch):
calls = []

Expand Down Expand Up @@ -486,6 +496,7 @@ def fake_varlen_func(q, k, v, **kwargs):
assert calls[0][3]["max_seqlen_q"] == 6


@pytest.mark.cpu
def test_packed_varlen_metadata_must_be_complete(monkeypatch):
monkeypatch.setattr(fa, "HAS_FLASH_ATTN", True)
impl = FlashAttentionImpl(num_heads=2, head_size=4, softmax_scale=0.5, causal=False)
Expand Down Expand Up @@ -542,6 +553,7 @@ def _fake_torch_npu(monkeypatch, *, npu_fusion_attention=None):
# --- Test group A: boundary resolution (_resolve_packed_seq_npu) -------------


@pytest.mark.cpu
def test_resolve_packed_seq_accepts_real_plus_pad_contract():
impl = _npu_impl()
q = torch.randn(1, 8, 2, 4)
Expand All @@ -551,6 +563,7 @@ def test_resolve_packed_seq_accepts_real_plus_pad_contract():
assert impl._resolve_packed_seq_npu(q, q, extra) == ([5, 8], [5, 8])


@pytest.mark.cpu
def test_resolve_packed_seq_single_document_no_padding():
impl = _npu_impl()
q = torch.randn(1, 8, 2, 4)
Expand All @@ -574,6 +587,7 @@ def test_resolve_packed_seq_single_document_no_padding():
"pad_longer_than_real",
],
)
@pytest.mark.cpu
def test_resolve_packed_seq_rejects_malformed_metadata(case):
"""Any metadata outside the exact [real, pad] contract returns None so the
caller falls back to the masked path."""
Expand Down Expand Up @@ -630,6 +644,7 @@ def _npu_impl_with_mocked_paths() -> tuple[FlashAttentionImpl, torch.Tensor]:
return impl, sentinel


@pytest.mark.cpu
def test_npu_env_unset_uses_varlen(monkeypatch):
monkeypatch.delenv("MINDIE_SD_FA_TYPE", raising=False)
_fake_mindiesd(monkeypatch)
Expand All @@ -642,6 +657,7 @@ def test_npu_env_unset_uses_varlen(monkeypatch):
impl._forward_prefix_kv_slice_npu.assert_not_called()


@pytest.mark.cpu
def test_npu_env_laser_uses_slice(monkeypatch):
monkeypatch.setenv("MINDIE_SD_FA_TYPE", "ascend_laser_attention")
_fake_mindiesd(monkeypatch)
Expand All @@ -655,6 +671,7 @@ def test_npu_env_laser_uses_slice(monkeypatch):


@pytest.mark.parametrize("fa_type", ["prompt_flash_attn", "fused_attn_score", "ascend_laser_attenton", "garbage"])
@pytest.mark.cpu
def test_npu_env_non_laser_value_falls_back_to_varlen(monkeypatch, fa_type):
"""Locks in the CURRENT dispatch behavior (PR #5891).

Expand All @@ -675,6 +692,7 @@ def test_npu_env_non_laser_value_falls_back_to_varlen(monkeypatch, fa_type):
impl._forward_prefix_kv_slice_npu.assert_not_called()


@pytest.mark.cpu
def test_npu_varlen_opt_in_unset_takes_mask_path(monkeypatch):
"""Models that do not set ``npu_attn_varlen`` (e.g. Wan/Cosmos) keep using
the existing masked attention_forward path; the new branches are skipped."""
Expand Down Expand Up @@ -702,6 +720,7 @@ def test_npu_varlen_opt_in_unset_takes_mask_path(monkeypatch):
("query_length", "key_length", "expected_inner_precise"),
[(4, 4, 0), (2, 4, 0), (4, 2, 2)],
)
@pytest.mark.cpu
def test_npu_causal_uses_native_right_down_mode(
monkeypatch,
query_length,
Expand Down Expand Up @@ -746,6 +765,7 @@ def test_npu_causal_uses_native_right_down_mode(
assert out is expected_out


@pytest.mark.cpu
def test_npu_causal_composes_explicit_keep_mask(monkeypatch):
expected_out = torch.randn(1, 4, 2, 4)
fusion_attention = Mock(return_value=(expected_out, None, None, None, 0, 0, 0))
Expand Down Expand Up @@ -782,6 +802,7 @@ def test_npu_causal_composes_explicit_keep_mask(monkeypatch):
assert out is expected_out


@pytest.mark.cpu
def test_npu_noncausal_without_explicit_mask_stays_unmasked(monkeypatch):
captured: dict = {}

Expand All @@ -800,6 +821,7 @@ def fake_attention_forward(query, key, value, **kwargs):
# --- Test group D: laser input pre-scaling in _forward_prefix_kv_slice_npu --


@pytest.mark.cpu
def test_prefix_kv_slice_applies_laser_input_scaling(monkeypatch):
monkeypatch.setenv("MINDIE_SD_FA_TYPE", "ascend_laser_attention")
captured: dict = {}
Expand Down Expand Up @@ -838,6 +860,7 @@ def fake_attention_forward(query, key, value, **kwargs):
torch.testing.assert_close(out, torch.ones(1, 8, 2, 4) * 256.0)


@pytest.mark.cpu
def test_prefix_kv_slice_no_scaling_without_factor(monkeypatch):
monkeypatch.setenv("MINDIE_SD_FA_TYPE", "ascend_laser_attention")
captured: dict = {}
Expand Down
3 changes: 2 additions & 1 deletion tests/diffusion/models/magi2/test_native_compile_cuda.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,12 +14,13 @@
)
from tests.helpers.mark import hardware_marks
from vllm_omni.diffusion.models.magi2.sampler_magi2 import CFGConfig
from vllm_omni.platforms import current_omni_platform

pytestmark = [
pytest.mark.diffusion,
pytest.mark.core_model,
*hardware_marks(res={"cuda": "L4"}, num_cards=1),
pytest.mark.skipif(not torch.cuda.is_available(), reason="requires CUDA"),
pytest.mark.skipif(not current_omni_platform.is_cuda(), reason="requires CUDA"),
]


Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -17,6 +17,7 @@
CausalHiFTGenerator,
HiFTGenerator,
)
from vllm_omni.platforms import current_omni_platform

pytestmark = [pytest.mark.core_model, pytest.mark.cpu]

Expand Down Expand Up @@ -735,6 +736,7 @@ def fake_stream_hift(self, feat, *, cache_state=None, finalize=False):


@pytest.mark.core_model
@pytest.mark.skipif(not current_omni_platform.is_cuda(), reason="requires CUDA")
@hardware_test(res={"cuda": "L4"}, num_cards=1)
def test_code2wav_streaming_batch_matches_ragged_flow_numerics(monkeypatch):
"""A padded flow batch must match individual flow calls on valid mels."""
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -31,6 +31,7 @@
_wrapped_slice,
)
from vllm_omni.model_executor.models.cosyvoice3.cosyvoice3_code2wav import CosyVoice3Code2Wav
from vllm_omni.platforms import current_omni_platform

pytestmark = [pytest.mark.core_model, pytest.mark.cpu]

Expand Down Expand Up @@ -334,6 +335,10 @@ def test_incremental_hift_matches_reference_for_voiced_f0(config):

@pytest.mark.parametrize("config", CONFIGS)
@pytest.mark.parametrize("first_len,step_len", [(10, 6), (10, 3), (20, 3), (10, 1)])
@pytest.mark.skipif(
current_omni_platform.is_rocm(),
reason="exhaustive CPU reference matrix exceeds the AMD CI time budget",
)
def test_incremental_hift_matches_reference_for_voiced_f0_small_chunks(config, first_len, step_len):
"""Same as test_incremental_hift_matches_reference_for_voiced_f0, but with small,
sub-receptive-field chunk sizes (down to 1 mel frame) instead of the uniform
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,7 @@
import pytest
import torch

from vllm_omni.platforms import current_omni_platform
from vllm_omni.model_executor.models.minicpmo_4_5.minicpmo_4_5_omni_llm import (
MiniCPMO45OmniLLMForConditionalGeneration,
)
Expand Down Expand Up @@ -38,7 +39,23 @@ def _reference_chunk_mask(
return mask


@pytest.mark.parametrize("size", [0, 1, 49, 50, 51, 1500])
@pytest.mark.parametrize(
"size",
[
0,
1,
49,
50,
51,
pytest.param(
1500,
marks=pytest.mark.skipif(
current_omni_platform.is_rocm(),
reason="large mask sizes exceed the AMD CI time budget",
),
),
],
)
@pytest.mark.parametrize("chunk_size", [1, 17, 50])
@pytest.mark.parametrize("num_left_chunks", [-1, 0, 1, 3])
@pytest.mark.parametrize("num_lookhead", [0, 1, 7])
Expand Down
Loading