Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
14 changes: 13 additions & 1 deletion tests/config/test_config_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,7 +8,8 @@

from vllm.config.cache import CacheConfig
from vllm.config.scheduler import SchedulerConfig
from vllm.config.utils import get_hash_factors, hash_factors, normalize_value
from vllm.config.utils import get_hash_factors, hash_factors, normalize_value, replace
from vllm.v1.kv_cache_layout import KVCacheLayout

# Helpers

Expand Down Expand Up @@ -217,6 +218,17 @@ def test_cache_config_hash_ignores_kv_cache_sizing_knobs():
assert CacheConfig(gpu_memory_utilization=0.5).compute_hash() == base_hash


def test_cache_config_replace_shares_resolved_kv_cache_layout():
target = CacheConfig(cache_dtype="auto")
draft = replace(target, cache_dtype="fp8")

target.kv_cache_layout = "LBHNC"

assert target.get_resolved_kv_cache_layout() is KVCacheLayout.LBHNC
assert draft.get_resolved_kv_cache_layout() is KVCacheLayout.LBHNC
assert target.metrics_info()["kv_cache_layout"] == "LBHNC"


def test_scheduler_config_hash_includes_max_num_seqs():
"""Per-request workspace sizes must invalidate compiled graphs."""
base_hash = SchedulerConfig(
Expand Down
32 changes: 27 additions & 5 deletions vllm/config/cache.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project

from collections.abc import Callable
from dataclasses import field
from dataclasses import dataclass, field
from functools import cache
from typing import Any, ClassVar, Literal

Expand Down Expand Up @@ -73,6 +73,13 @@ def _get_prefix_cache_retention_interval() -> int | None:
KVOffloadingBackend = Literal["native", "lmcache"]


@dataclass
class ResolvedKVCacheLayout:
"""Shared state for the resolved physical KV cache layout."""

value: str | None = None


@config
class CacheConfig:
"""Configuration for the KV cache."""
Expand All @@ -86,17 +93,29 @@ class CacheConfig:
"""Whether block_size was explicitly provided. Derived automatically."""
user_specified_mamba_block_size: bool = field(default=False, init=False)
"""Whether mamba_block_size was explicitly provided. Derived automatically."""
kv_cache_layout: str | None = field(default=None, init=False)
"""Resolved physical KV cache layout name (a ``KVCacheLayout`` member).
_resolved_kv_cache_layout: ResolvedKVCacheLayout = field(
default_factory=ResolvedKVCacheLayout,
repr=False,
)
"""Shared state for the resolved physical KV cache layout name.

``None`` means the layout has not been resolved yet. The engine core
A ``None`` value means the layout has not been resolved yet. The engine core
resolves it once (``resolve_kv_cache_layout``) before memory profiling, and
every worker process adopts the resolved name — via the
``set_kv_cache_layout`` RPC, or ``KVCacheConfig.kv_cache_layout`` for
workers spawned after resolution — before the KV cache is allocated. Once
set the value is final and read with ``get_resolved_kv_cache_layout``,
which raises on ``None``. Tests and standalone tools may pre-set a value,
which resolution then honors as-is."""

@property
def kv_cache_layout(self) -> str | None:
return self._resolved_kv_cache_layout.value

@kv_cache_layout.setter
def kv_cache_layout(self, value: str | None) -> None:
self._resolved_kv_cache_layout.value = value

prefix_match_unit: int | None = Field(default=None, gt=0)
"""The finest token boundary (in tokens) a prefix-cache hit can land on.

Expand Down Expand Up @@ -296,7 +315,10 @@ def compute_hash(self) -> str:
def metrics_info(self):
# convert cache_config to dict(key: str, value: str) for prometheus
# metrics info
return {key: str(value) for key, value in self.__dict__.items()}
info = {key: str(value) for key, value in self.__dict__.items()}
info.pop("_resolved_kv_cache_layout")
info["kv_cache_layout"] = str(self.kv_cache_layout)
return info

_block_size_resolved: bool = field(default=False, init=False)
"""Guard against pydantic re-running _apply_block_size_default."""
Expand Down
Loading