Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
56 changes: 53 additions & 3 deletions tests/v1/worker/test_gpu_model_runner_v2_cudagraph_profiling.py
Original file line number Diff line number Diff line change
Expand Up @@ -63,6 +63,7 @@ def _make_profiling_runner(
needs_capture, num_full_descs, piecewise_only
)
runner.vllm_config = SimpleNamespace()
runner.model_state = SimpleNamespace(supports_mm_inputs=False)

events: list[str] = []
runner.events = events
Expand Down Expand Up @@ -118,7 +119,8 @@ def _fake_set_current_vllm_config(_cfg):

def test_profile_cudagraph_memory_disabled_returns_zero(monkeypatch):
_patch_module(monkeypatch)
runner = _make_profiling_runner(CUDAGraphMode.NONE)
runner = _make_profiling_runner(CUDAGraphMode.NONE, needs_capture=False)
runner.cudagraph_manager = None

result = cgu.profile_cudagraph_memory(runner)

Expand All @@ -139,6 +141,34 @@ def test_profile_cudagraph_memory_no_graphs_tears_down(monkeypatch):
assert runner.cudagraph_manager.pool == GLOBAL_POOL


@pytest.mark.parametrize("mode", [CUDAGraphMode.NONE, CUDAGraphMode.FULL])
def test_profile_cudagraph_memory_accounts_for_encoder_without_decoder(
monkeypatch, mode
):
"""Encoder memory is budgeted even if the decoder has no graphs to capture."""
_patch_module(monkeypatch)
captured_bytes = 256 << 20
runner = _make_profiling_runner(
mode, needs_capture=False, captured_bytes=captured_bytes
)
runner.model_state = SimpleNamespace(
supports_mm_inputs=True,
encoder_runner=SimpleNamespace(has_cudagraph=lambda: True),
)
manager = runner.cudagraph_manager
runner.cudagraph_manager = None

def init(r):
r.events.append("init")
r.cudagraph_manager = manager

monkeypatch.setattr(cgu, "_init_minimal_kv_cache_for_profiling", init)

assert cgu.profile_cudagraph_memory(runner) == captured_bytes
assert runner.events == ["init", "capture", "teardown"]
assert _FakePlatform._global_graph_pool == GLOBAL_POOL


def test_profile_cudagraph_memory_samples_and_extrapolates(monkeypatch):
_patch_module(monkeypatch)
gib = 1 << 30
Expand All @@ -150,6 +180,14 @@ def test_profile_cudagraph_memory_samples_and_extrapolates(monkeypatch):
captured_bytes=1000 * gib,
mem_samples=[100 * gib, 20 * gib],
)
manager = runner.cudagraph_manager
runner.cudagraph_manager = None

def init(r):
r.events.append("init")
r.cudagraph_manager = manager

monkeypatch.setattr(cgu, "_init_minimal_kv_cache_for_profiling", init)

result = cgu.profile_cudagraph_memory(runner)

Expand Down Expand Up @@ -180,9 +218,20 @@ def test_profile_cudagraph_memory_piecewise_only_returns_measured(monkeypatch):
assert result == captured_bytes


def test_profile_cudagraph_memory_tears_down_on_capture_error(monkeypatch):
@pytest.mark.parametrize("encoder_only", [False, True])
def test_profile_cudagraph_memory_tears_down_on_capture_error(
monkeypatch, encoder_only
):
_patch_module(monkeypatch)
runner = _make_profiling_runner(CUDAGraphMode.FULL)
runner = _make_profiling_runner(
CUDAGraphMode.NONE if encoder_only else CUDAGraphMode.FULL,
needs_capture=not encoder_only,
)
if encoder_only:
runner.model_state = SimpleNamespace(
supports_mm_inputs=True,
encoder_runner=SimpleNamespace(has_cudagraph=lambda: True),
)

def _boom(*, profile_only: bool = False) -> int:
runner.events.append("capture")
Expand All @@ -199,6 +248,7 @@ def _boom(*, profile_only: bool = False) -> int:

# Teardown still runs even if capture raises.
assert runner.events == ["init", "capture", "teardown"]
assert _FakePlatform._global_graph_pool == GLOBAL_POOL


def test_profile_cudagraph_memory_restores_compilation_counters(monkeypatch):
Expand Down
75 changes: 74 additions & 1 deletion tests/v1/worker/test_gpu_worker.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,12 +3,13 @@

from contextlib import nullcontext
from types import SimpleNamespace
from unittest.mock import patch
from unittest.mock import Mock, patch

import pytest
import torch

from tests.utils import create_new_process_for_each_test
from vllm.config import CUDAGraphMode
from vllm.platforms import current_platform
from vllm.utils.mem_constants import GiB_bytes
from vllm.v1.worker import gpu_worker, startup_plan
Expand Down Expand Up @@ -243,6 +244,78 @@ def _profile_result(consumed, reserved_before=0, reserved_after=0):
)


@pytest.mark.skip_global_cleanup
@pytest.mark.parametrize("use_v2", [False, True], ids=["v1", "v2"])
@pytest.mark.parametrize("platform", ["cuda", "rocm", "xpu", "cpu"])
@pytest.mark.parametrize(
"decoder,encoder,eager,opt_in,profiled",
[
pytest.param(False, True, False, True, True, id="encoder-only"),
pytest.param(True, False, False, True, True, id="decoder-only"),
pytest.param(True, True, False, True, True, id="both"),
pytest.param(False, False, False, True, False, id="neither"),
pytest.param(False, True, True, True, False, id="enforce-eager"),
pytest.param(True, True, False, False, True, id="disabled-opt-in"),
],
)
def test_available_memory_reserves_enabled_cudagraphs(
monkeypatch, use_v2, platform, decoder, encoder, eager, opt_in, profiled
):
"""Profile either graph type, but reserve its estimate only with opt-in."""
monkeypatch.setenv("VLLM_ENABLE_STARTUP_PLAN", "0")
monkeypatch.setenv("VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS", str(int(opt_in)))
monkeypatch.setattr(
gpu_worker,
"current_platform",
SimpleNamespace(
is_cuda_alike=lambda: platform in ("cuda", "rocm"),
is_rocm=lambda: platform == "rocm",
is_xpu=lambda: platform == "xpu",
),
)
result = _profile_result(consumed=MEASURED_DROP)
result.non_kv_cache_memory = result.total_consumed
monkeypatch.setattr(
gpu_worker, "memory_profiling", lambda *args, **kwargs: nullcontext(result)
)
estimate = GiB_bytes
profile_graphs = Mock(return_value=estimate)
worker = SimpleNamespace(
_scoped_allocator_max_split=lambda **kwargs: nullcontext(),
use_v2_model_runner=use_v2,
vllm_config=SimpleNamespace(
compilation_config=SimpleNamespace(
cudagraph_mode=CUDAGraphMode.FULL if decoder else CUDAGraphMode.NONE,
cudagraph_mm_encoder=encoder,
),
),
cache_config=SimpleNamespace(
kv_cache_memory_bytes=None, gpu_memory_utilization=0.75
),
model_config=SimpleNamespace(multimodal_config=None, enforce_eager=eager),
parallel_config=SimpleNamespace(),
init_snapshot=SimpleNamespace(
free_memory=ANY_FREE_MEMORY, total_memory=ANY_FREE_MEMORY
),
requested_memory=6 * GiB_bytes,
model_runner=SimpleNamespace(
model_memory_usage=GiB_bytes,
profile_run=lambda: None,
profile_cudagraph_memory=profile_graphs,
),
)

available = gpu_worker.Worker.determine_available_memory(worker)

profiled = profiled and (decoder or use_v2) and platform != "cpu"
if profiled:
profile_graphs.assert_called_once_with()
else:
profile_graphs.assert_not_called()
reserved = estimate if profiled and opt_in else 0
assert available == worker.requested_memory - result.non_kv_cache_memory - reserved


@pytest.fixture
def rocm(request):
with patch.object(
Expand Down
7 changes: 5 additions & 2 deletions vllm/v1/worker/gpu/cudagraph_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -866,7 +866,10 @@ def profile_cudagraph_memory(runner: "GPUModelRunner") -> int:
partition reclaims the storages of earlier cudagraph recordings once the
real capture records new ones, leading to use-after-free crashes).
"""
if runner.compilation_config.cudagraph_mode == CUDAGraphMode.NONE:
if (
runner.compilation_config.cudagraph_mode == CUDAGraphMode.NONE
and not runner.needs_cudagraph_capture()
):
return 0

gc.collect()
Expand Down Expand Up @@ -901,7 +904,7 @@ def profile_cudagraph_memory(runner: "GPUModelRunner") -> int:
speculator = getattr(runner, "speculator", None)
spec_manager_names: list[str] = []
try:
if not manager.needs_capture():
if not runner.needs_cudagraph_capture():
return 0
manager.pool = throwaway_pool
if manager.use_breakable_cg:
Expand Down
12 changes: 9 additions & 3 deletions vllm/v1/worker/gpu_worker.py
Original file line number Diff line number Diff line change
Expand Up @@ -631,9 +631,15 @@ def determine_available_memory(self) -> int:
# the AMD-CI mem tests), and graph_pool_handle resolves to the same
# torch.cuda handle the live capture path already uses on ROCm.
cudagraph_memory_estimate = 0
if (
current_platform.is_cuda_alike() or current_platform.is_xpu()
) and self.vllm_config.compilation_config.cudagraph_mode != CUDAGraphMode.NONE:
compilation_config = self.vllm_config.compilation_config
if (current_platform.is_cuda_alike() or current_platform.is_xpu()) and (
compilation_config.cudagraph_mode != CUDAGraphMode.NONE
or (
self.use_v2_model_runner
and compilation_config.cudagraph_mm_encoder
and not self.model_config.enforce_eager
)
):
cudagraph_memory_estimate = self.model_runner.profile_cudagraph_memory()

# Respect the opt-in flag as originally designed.
Expand Down
Loading