Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
477 changes: 477 additions & 0 deletions tests/v1/attention/test_flashinfer_plan_from_bounds.py

Large diffs are not rendered by default.

8 changes: 8 additions & 0 deletions tests/v1/worker/test_gpu_ubatch_slicing.py
Original file line number Diff line number Diff line change
Expand Up @@ -121,6 +121,9 @@ def _make_input_batch(
query_start_loc_np=query_start_loc_np,
seq_lens=buffers.seq_lens[:num_reqs_padded],
seq_lens_cpu_upper_bound=torch.from_numpy(seq_lens_upper_bound),
seq_lens_cpu_lower_bound=torch.from_numpy(
np.maximum(seq_lens_upper_bound - 2, 0)
),
num_computed_tokens_np=np.array(seq_lens, dtype=np.int32)
- np.array(query_lens, dtype=np.int32),
prefill_len_np=np.zeros(num_reqs, dtype=np.int32),
Expand Down Expand Up @@ -187,6 +190,11 @@ def test_slicing_matches_v1_split_attn_metadata(batch_name: str):
torch.testing.assert_close(
v2_ubatch.seq_lens_cpu_upper_bound, v1_ubatch.seq_lens_cpu_upper_bound
)
# The lower bound is sliced and truncated like the upper bound.
torch.testing.assert_close(
v2_ubatch.seq_lens_cpu_lower_bound,
(v2_ubatch.seq_lens_cpu_upper_bound - 2).clamp(min=0),
)


def test_microbatches_do_not_share_buffers():
Expand Down
1 change: 1 addition & 0 deletions tests/v1/worker/test_gpu_warmup_blocks.py
Original file line number Diff line number Diff line change
Expand Up @@ -89,6 +89,7 @@ def _make_runner(
return SimpleNamespace(
num_speculative_steps=num_spec_steps,
adaptive_verification=None,
emit_seq_lens_cpu_lower_bound=True,
rejection_sampler=None,
decode_query_len=num_spec_steps + 1,
is_pooling_model=False,
Expand Down
2 changes: 2 additions & 0 deletions tests/v1/worker/test_mamba_hybrid_model_state.py
Original file line number Diff line number Diff line change
Expand Up @@ -43,6 +43,7 @@ def test_prepare_attn_forwards_positions(monkeypatch: pytest.MonkeyPatch) -> Non
num_scheduled_tokens=torch.tensor([1], dtype=torch.int32),
max_query_len=None,
seq_lens_cpu_upper_bound=torch.tensor([1537], dtype=torch.int32),
seq_lens_cpu_lower_bound=None,
seq_lens=torch.tensor([1537], dtype=torch.int32),
is_prefilling_np=torch.tensor([False]).numpy(),
prefill_runs_as_decode_np=None,
Expand Down Expand Up @@ -101,6 +102,7 @@ def test_padded_prompt_tail_builds_as_spec_decode(
num_draft_tokens_per_req=np.array([k, k, 0], dtype=np.int32),
max_query_len=None,
seq_lens_cpu_upper_bound=torch.tensor(seq_lens, dtype=torch.int32),
seq_lens_cpu_lower_bound=None,
seq_lens=torch.tensor(seq_lens, dtype=torch.int32),
is_prefilling_np=np.array(is_prefilling),
prefill_runs_as_decode_np=np.array([False, True, False]),
Expand Down
3 changes: 3 additions & 0 deletions tests/v1/worker/test_mixed_warmup_gate.py
Original file line number Diff line number Diff line change
Expand Up @@ -39,11 +39,13 @@ def test_kernel_warmup_restores_uncalibrated_adaptive_manager(monkeypatch, fail_
runner = SimpleNamespace(
adaptive_verification=manager,
rejection_sampler=rejection_sampler,
emit_seq_lens_cpu_lower_bound=True,
)

def run_steps(model_runner, execute, sample):
assert model_runner.adaptive_verification is None
assert not model_runner.rejection_sampler.enable_adaptive_verification
assert not model_runner.emit_seq_lens_cpu_lower_bound
if fail_warmup:
raise RuntimeError("warmup failed")

Expand All @@ -56,3 +58,4 @@ def run_steps(model_runner, execute, sample):
assert runner.adaptive_verification is manager
assert manager.cost_tables is None
assert rejection_sampler.enable_adaptive_verification
assert runner.emit_seq_lens_cpu_lower_bound
96 changes: 96 additions & 0 deletions tests/v1/worker/test_seq_len_bounds.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,96 @@
# SPDX-License-Identifier: Apache-2.0
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
"""The CPU bounds on seq_lens that FlashInfer plans from, and the check of the
exact seq_lens against them. No GPU needed."""

import numpy as np
import pytest
import torch

from vllm import envs
from vllm.v1.attention.backends.utils import check_seq_lens_bounds
from vllm.v1.worker.gpu.model_runner import compute_seq_lens_cpu_lower_bound
from vllm.v1.worker.gpu.spec_decode.speculator import (
compute_draft_seq_lens_cpu_lower_bound,
)

NUM_SPEC = 3
NUM_REQS = 6
# Six requests padded to eight. Row 3 is prefilling; rows 4 and 5 are the
# first tokens of a request, where the drafts in flight would go below zero.
UPPER = [1328, 21, 466, 65, 2, 3, 0, 0]
IS_PREFILLING = [False, False, False, True, False, False]
LOWER = [1325, 18, 463, 65, 0, 0, 0, 0]
EXACT = [1327, 20, 463, 65, 1, 3]


def test_runner_lower_bound_subtracts_the_drafts_in_flight() -> None:
upper_np = np.array(UPPER, dtype=np.int32)
lower_np = compute_seq_lens_cpu_lower_bound(
upper_np, np.array(IS_PREFILLING), NUM_SPEC, NUM_REQS
)
assert lower_np.tolist() == LOWER
assert lower_np.dtype == np.int32
# A new array; the upper bound is left as it was.
assert upper_np.tolist() == UPPER


def test_runner_lower_bound_without_drafts_is_the_upper_bound() -> None:
upper_np = np.array(UPPER, dtype=np.int32)
lower_np = compute_seq_lens_cpu_lower_bound(
upper_np, np.array(IS_PREFILLING), 0, NUM_REQS
)
assert lower_np.tolist() == UPPER
assert lower_np is not upper_np


@pytest.mark.parametrize(
"step, expected",
[
(1, [1323, 16, 461, 63, 0, 0]),
(2, [1324, 17, 462, 64, 0, 0]),
(3, [1325, 18, 463, 65, 0, 0]),
],
)
def test_draft_lower_bound_per_step(step: int, expected: list[int]) -> None:
"""Draft step ``step`` has ``step`` more tokens than the target lower bound
and up to NUM_SPEC fewer: the verification may have rejected that many."""
target_lower = torch.tensor(LOWER, dtype=torch.int32)
draft_lower = compute_draft_seq_lens_cpu_lower_bound(
target_lower, step, NUM_SPEC, NUM_REQS, num_reqs_padded=8
)
assert draft_lower.tolist() == expected + [0, 0]
assert draft_lower.dtype == torch.int32
assert target_lower.tolist() == LOWER


def _bounds() -> tuple[torch.Tensor, torch.Tensor]:
return (
torch.tensor(LOWER, dtype=torch.int32),
torch.tensor(UPPER, dtype=torch.int32),
)


def test_check_passes_within_the_bounds() -> None:
seq_lens = torch.tensor(EXACT, dtype=torch.int32)
# Longer bounds are cut to the requests; the bounds are inclusive.
check_seq_lens_bounds(seq_lens, *_bounds())
check_seq_lens_bounds(seq_lens, seq_lens, seq_lens)


@pytest.mark.parametrize(
"row, value",
[(0, 1329), (1, 17), (3, 64), (3, 66), (4, 3), (5, -1)],
)
def test_check_raises_outside_the_bounds(row: int, value: int) -> None:
seq_lens = torch.tensor(EXACT, dtype=torch.int32)
seq_lens[row] = value
with pytest.raises(RuntimeError):
check_seq_lens_bounds(seq_lens, *_bounds())


def test_check_is_off_by_default(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.delenv("VLLM_DEBUG_SEQ_LENS_BOUNDS", raising=False)
assert envs.VLLM_DEBUG_SEQ_LENS_BOUNDS is False
monkeypatch.setenv("VLLM_DEBUG_SEQ_LENS_BOUNDS", "1")
assert envs.VLLM_DEBUG_SEQ_LENS_BOUNDS is True
8 changes: 8 additions & 0 deletions vllm/envs.py
Original file line number Diff line number Diff line change
Expand Up @@ -299,6 +299,7 @@
VLLM_NCCL_INCLUDE_PATH: str | None = None
VLLM_GC_DEBUG: str = ""
VLLM_DEBUG_WORKSPACE: bool = False
VLLM_DEBUG_SEQ_LENS_BOUNDS: bool = False
VLLM_DISABLE_SHARED_EXPERTS_STREAM: bool = False
VLLM_DISABLE_DSV4_MEGAMOE_SHARED_EXPERT_FUSION: bool = False
VLLM_SHARED_EXPERTS_STREAM_TOKEN_THRESHOLD: int = 256
Expand Down Expand Up @@ -2059,6 +2060,13 @@ def _resolve_rust_cli_path() -> str | None:
# Debug workspace allocations.
# logging of workspace resize operations.
"VLLM_DEBUG_WORKSPACE": lambda: bool(int(os.getenv("VLLM_DEBUG_WORKSPACE", "0"))),
# Check on the device that the exact seq_lens lie within the CPU bounds
# FlashInfer plans from. Does not synchronize. Debug aid only: on CUDA a
# violation is a device-side assertion that leaves the CUDA context
# unusable. The FlashInfer builder reads this once at construction.
"VLLM_DEBUG_SEQ_LENS_BOUNDS": lambda: bool(
int(os.getenv("VLLM_DEBUG_SEQ_LENS_BOUNDS", "0"))
),
# Disables parallel execution of shared_experts via separate cuda stream
"VLLM_DISABLE_SHARED_EXPERTS_STREAM": lambda: bool(
int(os.getenv("VLLM_DISABLE_SHARED_EXPERTS_STREAM", "0"))
Expand Down
8 changes: 8 additions & 0 deletions vllm/v1/attention/backend.py
Original file line number Diff line number Diff line change
Expand Up @@ -454,6 +454,10 @@ class CommonAttentionMetadata:
table. Rows of one request are adjacent, so equal neighbours are PCP
chunks sharing one KV context."""

seq_lens_cpu_lower_bound: torch.Tensor | None = None
"""(batch_size,) CPU lower bound on seq_lens. It differs from the upper
bound only while drafts of a step still in flight may be rejected."""

mm_req_doc_ranges: dict[int, list[tuple[int, int]]] | None = None
"""PrefixLM bidirectional ranges for multimodal tokens. Maps
request index to list of (start, end) token position ranges
Expand Down Expand Up @@ -483,6 +487,9 @@ def naive_query_lens(self) -> torch.Tensor:
return self.query_start_loc[1:] - self.query_start_loc[:-1]

def replace(self, **kwargs) -> "CommonAttentionMetadata":
if "seq_lens_cpu_upper_bound" in kwargs or "seq_lens" in kwargs:
# A lower bound not updated along with seq_lens is stale.
kwargs.setdefault("seq_lens_cpu_lower_bound", None)
return replace(self, **kwargs)

def compute_num_computed_tokens(self) -> torch.Tensor:
Expand Down Expand Up @@ -549,6 +556,7 @@ def unpadded(
self.dcp_local_seq_lens_cpu_upper_bound
),
seq_lens_cpu_upper_bound=maybe_slice_reqs(self.seq_lens_cpu_upper_bound),
seq_lens_cpu_lower_bound=maybe_slice_reqs(self.seq_lens_cpu_lower_bound),
is_prefilling=maybe_slice_reqs(self.is_prefilling),
req_idx=maybe_slice_reqs(self.req_idx),
rswa_prefix_lens=maybe_slice_reqs(self.rswa_prefix_lens),
Expand Down
Loading