From 8c75d67b9dcb76fa4c05d2e4db62ea499fcacaec Mon Sep 17 00:00:00 2001 From: BWAAEEEK Date: Tue, 22 Sep 2026 07:20:10 +0000 Subject: [PATCH] [Bugfix][MRV2] Reserve encoder graph memory without decoder graphs Allow the worker and V2 graph profiler to account for encoder CUDA graphs when decoder graphs are disabled. Keep the pre-bootstrap decoder check, MRV1 behavior, eager exclusion, and graph-reservation opt-out semantics. Add regression coverage for profiling eligibility, KV budget subtraction, and cleanup when encoder graph capture fails. Assisted-by: OpenAI Codex Signed-off-by: BWAAEEEK --- ...gpu_model_runner_v2_cudagraph_profiling.py | 56 +++++++++++++- tests/v1/worker/test_gpu_worker.py | 75 ++++++++++++++++++- vllm/v1/worker/gpu/cudagraph_utils.py | 7 +- vllm/v1/worker/gpu_worker.py | 12 ++- 4 files changed, 141 insertions(+), 9 deletions(-) diff --git a/tests/v1/worker/test_gpu_model_runner_v2_cudagraph_profiling.py b/tests/v1/worker/test_gpu_model_runner_v2_cudagraph_profiling.py index 74191206d363..7bb5c70bf6cf 100644 --- a/tests/v1/worker/test_gpu_model_runner_v2_cudagraph_profiling.py +++ b/tests/v1/worker/test_gpu_model_runner_v2_cudagraph_profiling.py @@ -63,6 +63,7 @@ def _make_profiling_runner( needs_capture, num_full_descs, piecewise_only ) runner.vllm_config = SimpleNamespace() + runner.model_state = SimpleNamespace(supports_mm_inputs=False) events: list[str] = [] runner.events = events @@ -118,7 +119,8 @@ def _fake_set_current_vllm_config(_cfg): def test_profile_cudagraph_memory_disabled_returns_zero(monkeypatch): _patch_module(monkeypatch) - runner = _make_profiling_runner(CUDAGraphMode.NONE) + runner = _make_profiling_runner(CUDAGraphMode.NONE, needs_capture=False) + runner.cudagraph_manager = None result = cgu.profile_cudagraph_memory(runner) @@ -139,6 +141,34 @@ def test_profile_cudagraph_memory_no_graphs_tears_down(monkeypatch): assert runner.cudagraph_manager.pool == GLOBAL_POOL +@pytest.mark.parametrize("mode", [CUDAGraphMode.NONE, CUDAGraphMode.FULL]) +def test_profile_cudagraph_memory_accounts_for_encoder_without_decoder( + monkeypatch, mode +): + """Encoder memory is budgeted even if the decoder has no graphs to capture.""" + _patch_module(monkeypatch) + captured_bytes = 256 << 20 + runner = _make_profiling_runner( + mode, needs_capture=False, captured_bytes=captured_bytes + ) + runner.model_state = SimpleNamespace( + supports_mm_inputs=True, + encoder_runner=SimpleNamespace(has_cudagraph=lambda: True), + ) + manager = runner.cudagraph_manager + runner.cudagraph_manager = None + + def init(r): + r.events.append("init") + r.cudagraph_manager = manager + + monkeypatch.setattr(cgu, "_init_minimal_kv_cache_for_profiling", init) + + assert cgu.profile_cudagraph_memory(runner) == captured_bytes + assert runner.events == ["init", "capture", "teardown"] + assert _FakePlatform._global_graph_pool == GLOBAL_POOL + + def test_profile_cudagraph_memory_samples_and_extrapolates(monkeypatch): _patch_module(monkeypatch) gib = 1 << 30 @@ -150,6 +180,14 @@ def test_profile_cudagraph_memory_samples_and_extrapolates(monkeypatch): captured_bytes=1000 * gib, mem_samples=[100 * gib, 20 * gib], ) + manager = runner.cudagraph_manager + runner.cudagraph_manager = None + + def init(r): + r.events.append("init") + r.cudagraph_manager = manager + + monkeypatch.setattr(cgu, "_init_minimal_kv_cache_for_profiling", init) result = cgu.profile_cudagraph_memory(runner) @@ -180,9 +218,20 @@ def test_profile_cudagraph_memory_piecewise_only_returns_measured(monkeypatch): assert result == captured_bytes -def test_profile_cudagraph_memory_tears_down_on_capture_error(monkeypatch): +@pytest.mark.parametrize("encoder_only", [False, True]) +def test_profile_cudagraph_memory_tears_down_on_capture_error( + monkeypatch, encoder_only +): _patch_module(monkeypatch) - runner = _make_profiling_runner(CUDAGraphMode.FULL) + runner = _make_profiling_runner( + CUDAGraphMode.NONE if encoder_only else CUDAGraphMode.FULL, + needs_capture=not encoder_only, + ) + if encoder_only: + runner.model_state = SimpleNamespace( + supports_mm_inputs=True, + encoder_runner=SimpleNamespace(has_cudagraph=lambda: True), + ) def _boom(*, profile_only: bool = False) -> int: runner.events.append("capture") @@ -199,6 +248,7 @@ def _boom(*, profile_only: bool = False) -> int: # Teardown still runs even if capture raises. assert runner.events == ["init", "capture", "teardown"] + assert _FakePlatform._global_graph_pool == GLOBAL_POOL def test_profile_cudagraph_memory_restores_compilation_counters(monkeypatch): diff --git a/tests/v1/worker/test_gpu_worker.py b/tests/v1/worker/test_gpu_worker.py index b63681deca2c..5fa80b01570f 100644 --- a/tests/v1/worker/test_gpu_worker.py +++ b/tests/v1/worker/test_gpu_worker.py @@ -3,12 +3,13 @@ from contextlib import nullcontext from types import SimpleNamespace -from unittest.mock import patch +from unittest.mock import Mock, patch import pytest import torch from tests.utils import create_new_process_for_each_test +from vllm.config import CUDAGraphMode from vllm.platforms import current_platform from vllm.utils.mem_constants import GiB_bytes from vllm.v1.worker import gpu_worker, startup_plan @@ -243,6 +244,78 @@ def _profile_result(consumed, reserved_before=0, reserved_after=0): ) +@pytest.mark.skip_global_cleanup +@pytest.mark.parametrize("use_v2", [False, True], ids=["v1", "v2"]) +@pytest.mark.parametrize("platform", ["cuda", "rocm", "xpu", "cpu"]) +@pytest.mark.parametrize( + "decoder,encoder,eager,opt_in,profiled", + [ + pytest.param(False, True, False, True, True, id="encoder-only"), + pytest.param(True, False, False, True, True, id="decoder-only"), + pytest.param(True, True, False, True, True, id="both"), + pytest.param(False, False, False, True, False, id="neither"), + pytest.param(False, True, True, True, False, id="enforce-eager"), + pytest.param(True, True, False, False, True, id="disabled-opt-in"), + ], +) +def test_available_memory_reserves_enabled_cudagraphs( + monkeypatch, use_v2, platform, decoder, encoder, eager, opt_in, profiled +): + """Profile either graph type, but reserve its estimate only with opt-in.""" + monkeypatch.setenv("VLLM_ENABLE_STARTUP_PLAN", "0") + monkeypatch.setenv("VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS", str(int(opt_in))) + monkeypatch.setattr( + gpu_worker, + "current_platform", + SimpleNamespace( + is_cuda_alike=lambda: platform in ("cuda", "rocm"), + is_rocm=lambda: platform == "rocm", + is_xpu=lambda: platform == "xpu", + ), + ) + result = _profile_result(consumed=MEASURED_DROP) + result.non_kv_cache_memory = result.total_consumed + monkeypatch.setattr( + gpu_worker, "memory_profiling", lambda *args, **kwargs: nullcontext(result) + ) + estimate = GiB_bytes + profile_graphs = Mock(return_value=estimate) + worker = SimpleNamespace( + _scoped_allocator_max_split=lambda **kwargs: nullcontext(), + use_v2_model_runner=use_v2, + vllm_config=SimpleNamespace( + compilation_config=SimpleNamespace( + cudagraph_mode=CUDAGraphMode.FULL if decoder else CUDAGraphMode.NONE, + cudagraph_mm_encoder=encoder, + ), + ), + cache_config=SimpleNamespace( + kv_cache_memory_bytes=None, gpu_memory_utilization=0.75 + ), + model_config=SimpleNamespace(multimodal_config=None, enforce_eager=eager), + parallel_config=SimpleNamespace(), + init_snapshot=SimpleNamespace( + free_memory=ANY_FREE_MEMORY, total_memory=ANY_FREE_MEMORY + ), + requested_memory=6 * GiB_bytes, + model_runner=SimpleNamespace( + model_memory_usage=GiB_bytes, + profile_run=lambda: None, + profile_cudagraph_memory=profile_graphs, + ), + ) + + available = gpu_worker.Worker.determine_available_memory(worker) + + profiled = profiled and (decoder or use_v2) and platform != "cpu" + if profiled: + profile_graphs.assert_called_once_with() + else: + profile_graphs.assert_not_called() + reserved = estimate if profiled and opt_in else 0 + assert available == worker.requested_memory - result.non_kv_cache_memory - reserved + + @pytest.fixture def rocm(request): with patch.object( diff --git a/vllm/v1/worker/gpu/cudagraph_utils.py b/vllm/v1/worker/gpu/cudagraph_utils.py index 8f8658077c13..5ada2816ca94 100644 --- a/vllm/v1/worker/gpu/cudagraph_utils.py +++ b/vllm/v1/worker/gpu/cudagraph_utils.py @@ -866,7 +866,10 @@ def profile_cudagraph_memory(runner: "GPUModelRunner") -> int: partition reclaims the storages of earlier cudagraph recordings once the real capture records new ones, leading to use-after-free crashes). """ - if runner.compilation_config.cudagraph_mode == CUDAGraphMode.NONE: + if ( + runner.compilation_config.cudagraph_mode == CUDAGraphMode.NONE + and not runner.needs_cudagraph_capture() + ): return 0 gc.collect() @@ -901,7 +904,7 @@ def profile_cudagraph_memory(runner: "GPUModelRunner") -> int: speculator = getattr(runner, "speculator", None) spec_manager_names: list[str] = [] try: - if not manager.needs_capture(): + if not runner.needs_cudagraph_capture(): return 0 manager.pool = throwaway_pool if manager.use_breakable_cg: diff --git a/vllm/v1/worker/gpu_worker.py b/vllm/v1/worker/gpu_worker.py index def92f601dc8..b8889af25038 100644 --- a/vllm/v1/worker/gpu_worker.py +++ b/vllm/v1/worker/gpu_worker.py @@ -631,9 +631,15 @@ def determine_available_memory(self) -> int: # the AMD-CI mem tests), and graph_pool_handle resolves to the same # torch.cuda handle the live capture path already uses on ROCm. cudagraph_memory_estimate = 0 - if ( - current_platform.is_cuda_alike() or current_platform.is_xpu() - ) and self.vllm_config.compilation_config.cudagraph_mode != CUDAGraphMode.NONE: + compilation_config = self.vllm_config.compilation_config + if (current_platform.is_cuda_alike() or current_platform.is_xpu()) and ( + compilation_config.cudagraph_mode != CUDAGraphMode.NONE + or ( + self.use_v2_model_runner + and compilation_config.cudagraph_mm_encoder + and not self.model_config.enforce_eager + ) + ): cudagraph_memory_estimate = self.model_runner.profile_cudagraph_memory() # Respect the opt-in flag as originally designed.