Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
53 changes: 53 additions & 0 deletions vllm_ascend/_310p/worker_310p.py
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,7 @@
import psutil
import torch
import torch_npu
from vllm.config import CUDAGraphMode
from vllm.logger import logger
from vllm.utils.mem_constants import GiB_bytes
from vllm.utils.mem_utils import MemorySnapshot, memory_profiling
Expand All @@ -29,6 +30,7 @@
from vllm_ascend.worker.worker import NPUWorker, init_workspace_manager

_IS_RC_DEVICE: bool | None = None
_FP16_BYTES = 2


def _is_rc_device() -> bool:
Expand All @@ -45,6 +47,45 @@ def _is_rc_device() -> bool:
return _IS_RC_DEVICE


def _get_310p_aclgraph_memory_reserve(vllm_config) -> int:
"""Estimate extra memory to leave free for 310P ACL graph capture.

310P currently materializes full fp16 causal masks and casts them to
FRACTAL_NZ. During ACL graph capture, these mask/workspace allocations can
overlap with graph-pool allocations, so the KV cache budget needs a small
device-specific guard band.
"""
compilation_config = vllm_config.compilation_config
if compilation_config.cudagraph_mode == CUDAGraphMode.NONE:
return 0

additional_config = vllm_config.additional_config or {}
if "aclgraph_memory_reserve_bytes" in additional_config:
return int(additional_config["aclgraph_memory_reserve_bytes"])
if "aclgraph_memory_reserve_gib" in additional_config:
return int(float(additional_config["aclgraph_memory_reserve_gib"]) * GiB_bytes)

capture_sizes = compilation_config.cudagraph_capture_sizes or []
if not capture_sizes:
return 0

max_model_len = vllm_config.model_config.max_model_len
if max_model_len <= 0:
return GiB_bytes
Comment on lines +72 to +74

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

high

In some testing or dummy initialization scenarios, vllm_config.model_config can be None. Accessing vllm_config.model_config.max_model_len directly without a guard can lead to an AttributeError. We should safely retrieve model_config first and default max_model_len to 0 if it is not present.

Suggested change
max_model_len = vllm_config.model_config.max_model_len
if max_model_len <= 0:
return GiB_bytes
model_config = getattr(vllm_config, "model_config", None)
max_model_len = model_config.max_model_len if model_config else 0
if max_model_len <= 0:
return GiB_bytes


full_mask_bytes = max_model_len * max_model_len * _FP16_BYTES

# The cached full causal mask is usually materialized during profile_run
# and should already be reflected in non-KV memory. Reserve one mask-sized
# guard for FRACTAL_NZ conversion/transient workspace, then add a small
# per-graph allowance for graph-pool overhead.
per_graph_overhead = 512 * (1 << 20)
estimated_reserve = full_mask_bytes + len(capture_sizes) * per_graph_overhead
min_reserve = GiB_bytes
max_reserve = 4 * GiB_bytes
return int(min(max(estimated_reserve, min_reserve), max_reserve))


class NPUWorker310(NPUWorker):
def init_device(self):
self.device = self._init_device()
Expand Down Expand Up @@ -128,6 +169,18 @@ def determine_available_memory(self) -> int:
self.requested_memory - profile_result.non_kv_cache_memory - non_torch_memory_cleared_by_empty_cache
) // 2

aclgraph_memory_reserve = _get_310p_aclgraph_memory_reserve(self.vllm_config)
if aclgraph_memory_reserve > 0:
self.available_kv_cache_memory_bytes = max(
0,
self.available_kv_cache_memory_bytes - aclgraph_memory_reserve,
)
logger.info_once(
"Reserved %.2f GiB from 310P KV cache budget for ACL graph capture mask/workspace memory.",
GiB(aclgraph_memory_reserve),
scope="local",
)

logger.debug(profile_result)
logger.info_once(
"Available KV cache memory: %.2f GiB",
Expand Down
Loading
Loading