Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
173 commits
Select commit Hold shift + click to select a range
22b209a
Support unified memory with speculative decoding
Aug 26, 2026
e887a18
fix: use hybrid SWA draft flag for pool setup
merrymercy Sep 1, 2026
ad6c3b4
Use shared byte capacity for unified SWA pools
Aug 27, 2026
1aac878
Avoid double-rounding unified SWA demand
ZYHowell Aug 27, 2026
7ee2c74
Reject unsupported unified pool scheduler modes
Aug 27, 2026
c54b97d
Remove unified pool gate-only tests
Aug 28, 2026
8892c52
Limit shared-capacity changes to enabled paths
Aug 28, 2026
48b2a24
Fix unified SWA eviction over-reclaim
Aug 28, 2026
139a9a1
Remove added unified eviction planner test
YhZhuang Aug 28, 2026
9c31d3d
Remove low-signal unified capacity tests
Sep 1, 2026
1599357
Adapt shared capacity to current main
Sep 1, 2026
c2296be
Address unified SWA admission feedback
Sep 1, 2026
05cfb81
Support unified memory page envelopes in PD transfers
Aug 27, 2026
c4551a5
Use decode request pool for unified SWA PD
Aug 27, 2026
623f762
Reject unified Mooncake speculative transfers
Aug 27, 2026
c9b4c29
Simplify unified draft index plumbing
Aug 28, 2026
5a7d1ef
Fix unified SWA PD capacity accounting
Sep 1, 2026
d5e5e7a
Support unified memory decode host pools
Aug 27, 2026
705f648
Fix unified draft retraction indices
Aug 28, 2026
f37d710
Simplify unified decode host allocation
Aug 28, 2026
1b75db1
Gate unsupported unified hybrid host transfers
Sep 2, 2026
f806ad5
Fix unified HiCache physical transfers
Sep 2, 2026
08a06eb
Prefer same-side eviction for unified SWA
ZYHowell Sep 2, 2026
6c46fed
Merge updated shared-capacity branch into PD page envelopes
ZYHowell Sep 2, 2026
28962eb
Merge updated PD page-envelope branch into decode host pool
ZYHowell Sep 2, 2026
fe5787a
Prefer SWA tombstones before unified FULL eviction
Sep 2, 2026
5d19c4f
Merge updated shared-capacity policy
Sep 2, 2026
bdb7287
Merge updated UMP PD stack into host pool
Sep 2, 2026
e6e491b
Restore upstream unified cache eviction policy
Sep 2, 2026
cc12f23
Merge restored eviction policy into PD page envelopes
Sep 2, 2026
45b4663
Merge restored eviction policy into decode host pool
Sep 2, 2026
66fed93
Use joint capacity accounting for unified SWA
Sep 2, 2026
aa6d1cd
Merge joint capacity accounting into PD page envelopes
Sep 2, 2026
43ebdb4
Merge joint capacity accounting into decode host pool
Sep 2, 2026
329fb0c
Merge latest unified host-pool stack
Sep 2, 2026
db64894
Refactor unified host pool allocation
Sep 4, 2026
f16293b
Support unified HiCache buffer staging
Sep 4, 2026
54d3892
Merge compacting unified host pool
Sep 4, 2026
d3e06da
Reject unsafe unified decode host paths
Sep 4, 2026
136669c
Enable shared unified HiCache staging safely
Sep 4, 2026
63abe66
Merge unified decode host safety gates
Sep 4, 2026
f0af6be
Guard unified HiCache compatibility
Sep 4, 2026
c41b068
Add unified capacity regression coverage
Sep 4, 2026
43ecd06
Use physical SWA indices for PD transfer
Sep 4, 2026
6fe1e7d
Merge unified capacity fixes into PD envelopes
Sep 4, 2026
3ef84f1
Merge transfer-index fixes into decode host pool
Sep 4, 2026
1f8c4fe
Merge review fixes into HiCache transfers
Sep 4, 2026
d315b7d
Merge remote-tracking branch 'origin/main' into yonghao/ump-shared-ca…
Sep 4, 2026
7fadf8e
Merge commit 'd315b7d65e' into yonghao/ump-pd-page-envelopes
Sep 4, 2026
372d211
Merge commit '7fadf8efc0' into yonghao/ump-decode-host-pool
Sep 4, 2026
d4462ed
Merge commit '372d2111cc' into yonghao/ump-hicache-physical-transfers
Sep 4, 2026
940d41b
Update PD priority test fixture for draft indices
Sep 4, 2026
d01513d
Update HiSparse test fixture for unified memory
Sep 4, 2026
25d1e3b
Merge commit 'd01513db8e' into yonghao/ump-pd-page-envelopes
Sep 4, 2026
ec0e777
Merge commit '25d1e3bc64d22be2ed3665e033a2218a1f7cbce2' into yonghao/…
Sep 4, 2026
f417938
Merge commit 'ec0e7777f38d651b0c3e849d033b81985fc091b3' into yonghao/…
Sep 4, 2026
b0018de
Update decode offload test fixtures for sidecars
Sep 4, 2026
3c26277
Merge commit 'b0018def8d' into yonghao/ump-hicache-physical-transfers
Sep 4, 2026
18142d6
Keep physical reservation cleanup host-sync-free
Sep 4, 2026
8bc1586
Plan unified FULL/SWA reclaim against physical capacity
Sep 8, 2026
7f10ff8
Merge shared-capacity fixes into page-envelope transfers
Sep 8, 2026
6f9a8f2
Merge capacity fixes and guard shared host-layout access
Sep 8, 2026
95e9b6d
Merge host-pool updates and preserve unified transfer lifetimes
Sep 8, 2026
653c010
Restore unified pool cleanup and scoped compatibility gates
Sep 8, 2026
07df3f4
Merge shared-capacity cleanup into PD page envelopes
Sep 8, 2026
d9ce0bb
Merge UMP cleanup into shared host pool
Sep 8, 2026
097e185
Merge UMP cleanup and restore host-sync-free physical release
Sep 8, 2026
f83adaf
Merge upstream main and adapt unified capacity to allocator refactors
Sep 10, 2026
b32e13e
Merge shared capacity refresh and adapt PD envelope allocators
Sep 10, 2026
a3a32f2
Merge PD refresh and reconcile shared host pool with upstream UMBP
Sep 10, 2026
b20d10f
Merge host pool refresh and adapt HiCache physical transfers
Sep 10, 2026
f604efb
Keep HiSparse fixture compatible with Python 3.10
Sep 10, 2026
e09999e
Merge Python 3.10 fixture compatibility fix from shared capacity
Sep 10, 2026
35519a8
Merge Python 3.10 fixture compatibility fix from PD envelopes
Sep 10, 2026
87df0bc
Merge Python 3.10 fixture compatibility fix from host pool
Sep 10, 2026
70312b5
Remove unrelated UMP diff churn
Sep 10, 2026
054c220
Merge shared-capacity cleanup into PD page envelopes
Sep 10, 2026
7a71192
Merge PD cleanup into decode host pool
Sep 10, 2026
709e1df
Merge host-pool cleanup into physical transfers
Sep 10, 2026
4e0e1c3
Fix unified pool admission rounding
Sep 12, 2026
69747f6
Fix full-envelope PD transfers and merge capacity fixes
Sep 12, 2026
b7c0f51
Fix L2 producer ordering and host element metadata
Sep 12, 2026
3870148
Fix SWA-only backup and shared-host prefetch fallback
Sep 12, 2026
70671d5
Refactor unified SWA budget handling
Sep 13, 2026
66a2172
Merge shared-capacity cleanup
Sep 13, 2026
f7cda39
Merge PD shared-capacity cleanup
Sep 13, 2026
b03bcd2
Merge host-pool shared-capacity cleanup
Sep 13, 2026
cb88f1f
Centralize decode SWA reservation routing
Sep 13, 2026
415c3bc
Merge decode reservation cleanup
Sep 13, 2026
759b428
Unify host transfer lifecycle and index preparation
Sep 13, 2026
dc84a3b
Merge shared host transfer cleanup
Sep 13, 2026
320d041
Separate storage hit allocation from queue policy
Sep 13, 2026
809821e
Keep short-prefix threshold policy at the cache boundary
Sep 13, 2026
29545f8
Simplify shared SWA budget paths
Sep 13, 2026
8d4460c
Merge shared SWA budget cleanup
Sep 13, 2026
3b85f00
Merge shared SWA budget cleanup
Sep 13, 2026
242578b
Merge shared SWA budget cleanup
Sep 13, 2026
4d1fa0f
Fix unified capturer indexing and asymmetric SWA reservation
Sep 13, 2026
d0a10e5
Merge shared-capacity fixes and repair unified CPU copy API
Sep 13, 2026
4ed9263
Merge PD fixes and use physical SWA offload indices
Sep 13, 2026
e951bb1
Merge host-pool fixes and preserve eviction progress under host pressure
Sep 13, 2026
567f4f5
Consolidate unified allocation, admission and capacity policies
Sep 13, 2026
6066493
Merge capacity cleanup and cover real SWA tail allocation
Sep 13, 2026
1e3a2c4
Merge unified admission and PD allocation review cleanup
Sep 13, 2026
1fcd125
Merge UMP cleanup and preserve non-unified HiCache failure semantics
Sep 13, 2026
6c7be62
Refactor SWA admission into allocator-owned memory budgets
ch-wan Sep 13, 2026
4fdb2a7
Fix unaligned unified SWA prefill admission
Sep 14, 2026
5a8f410
Merge unified prefill rounding fix into PD support
Sep 14, 2026
39a7bcc
Merge unified prefill rounding fix into host pool
Sep 14, 2026
a9a744a
Merge unified prefill rounding fix into HiCache transfers
Sep 14, 2026
089673a
Refactor unified SWA allocators into sibling capacity policies
Sep 14, 2026
121fdde
Clean up unified PD rejection and transfer interfaces
Sep 14, 2026
1ff144d
Unify host-pool allocation and L2 transfer lifecycles
Sep 14, 2026
e843a0a
Consolidate UMBP page layout and SWA host-pool dispatch
Sep 14, 2026
3c9040a
Merge reviewed shared-capacity cleanup into PD layer
Sep 14, 2026
a8e0085
Merge reviewed PD cleanup into host-pool layer
Sep 14, 2026
8e0df01
Merge host-pool cleanup preserving HiCache bindings
Sep 14, 2026
564845d
Use shared byte capacity for unified SWA pools
Sep 14, 2026
a7084b3
Keep HiSparse fixture compatible with Python 3.10
Sep 10, 2026
b505c11
Remove unrelated UMP diff churn
Sep 10, 2026
99ee780
Fix unified pool admission rounding
Sep 12, 2026
8fd4aa4
Refactor unified SWA budget handling
Sep 13, 2026
fa6c95a
Simplify shared SWA budget paths
Sep 13, 2026
d9c7447
Fix unified capturer indexing and asymmetric SWA reservation
Sep 13, 2026
6cf450c
Consolidate unified allocation, admission and capacity policies
Sep 13, 2026
0a80658
Refactor SWA admission into allocator-owned memory budgets
ch-wan Sep 13, 2026
1f3b496
Fix unaligned unified SWA prefill admission
Sep 14, 2026
f208606
Refactor unified SWA allocators into sibling capacity policies
Sep 14, 2026
f0a0085
Fix shared prefill deferral and capture and draft capacity bounds
ch-wan Sep 14, 2026
ab2ea29
Validate inherited unified allocator PD and free-group contracts
ch-wan Sep 14, 2026
e3fa99e
Merge updated shared-capacity branch into UMP PD support
Sep 14, 2026
5e5433a
Merge updated UMP PD support into unified host pools
Sep 14, 2026
f7f0741
Update staged transfer fixtures for host-pool hooks
Sep 14, 2026
74b7f42
Merge updated unified host pools and preserve HiCache transfer fixes
Sep 14, 2026
ab7d973
Explain SWA prefill reservation and chunking safety
Sep 14, 2026
ea634ec
Merge shared-capacity updates and simplify PD reservation checks
Sep 14, 2026
eb907b6
Merge PD updates and unify host layout and write interfaces
Sep 14, 2026
acce638
Merge host-pool updates and simplify HiCache allocation policies
Sep 14, 2026
e6a273f
Use declared HiCache transfer capability fields
Sep 14, 2026
ccd3487
Fix Rust SWA backup sources and gate sidecar error recovery
Sep 14, 2026
d9c9565
Merge host-pool capability cleanup into HiCache transfers
Sep 14, 2026
81d00a3
Refresh shared KV budgets during decode admission
Sep 14, 2026
a640ec3
Align HiCache prefix reuse with SWA tail allocation
Sep 14, 2026
405e92b
Port profiled SWA byte-budget fix from #36729
Sep 14, 2026
bd62e85
Merge shared KV budget fixes into unified host pool
Sep 14, 2026
bbbebd7
Merge shared KV budget fixes into HiCache physical transfers
Sep 14, 2026
1c2fe92
Merge latest main after shared-capacity PR landed
ZYHowell Sep 15, 2026
3606f52
Merge updated PD page-envelope base
ZYHowell Sep 15, 2026
ba4a021
Sync main through b803cfa0c4
ZYHowell Sep 15, 2026
7373a45
Sync updated PD base through latest main
ZYHowell Sep 15, 2026
3e1d4de
Merge updated host-pool base and resolve HiCache conflicts
ZYHowell Sep 15, 2026
15ec3e5
Sync updated host-pool base through latest main
ZYHowell Sep 15, 2026
f7e5443
Share precise SWA tail allocation across unified layouts
ZYHowell Sep 16, 2026
4278c21
Merge shared SWA tail allocation fix from lower stack
ZYHowell Sep 16, 2026
7269c5c
Merge shared SWA tail allocation fix from lower stack
ZYHowell Sep 16, 2026
e514294
Merge main b02e16a895 and retain both PD transfer test groups
ZYHowell Sep 16, 2026
2281f9a
Merge fixed main snapshot through the PD base
ZYHowell Sep 16, 2026
1e1c33d
Merge fixed main snapshot and preserve HiCache prefix metadata with e…
ZYHowell Sep 16, 2026
086fe1a
Merge main and reconcile unified HiCache transfer contracts
Sep 24, 2026
5f17c46
Merge latest main and retain focused component-only fallback coverage
Sep 24, 2026
e8717bb
Merge remote-tracking branch 'origin/main' into task/t-0010-pr39479
Sep 24, 2026
a213b8a
Preserve Mooncake buffer host pools and reconcile allocator test cont…
Sep 24, 2026
64eda39
Clean redundant unified HiCache test setup
Sep 28, 2026
6027473
Align unified HiCache storage and staged SWA ownership
Sep 29, 2026
7b79095
Remove redundant shared-host allocation collectives
Sep 30, 2026
4154c31
Align HiCache host refill and validate write policy transitions
Sep 30, 2026
573ff05
Document incoming host rows retained by tree refill
Sep 30, 2026
95adcf7
Merge main into unified memory HiCache transfers
ZYHowell Oct 1, 2026
b55790f
Preserve unified cache transfer lifetimes and SWA physical indices
ZYHowell Oct 1, 2026
e8364fe
Merge main into unified-memory HiCache transfers
ZYHowell Oct 2, 2026
2ce5783
Merge main into unified-memory HiCache transfers
ZYHowell Oct 3, 2026
90e9dcb
Merge main into unified-memory HiCache transfers (resolve _set_capaci…
metamergebot Oct 4, 2026
e6487f4
Merge main into unified-memory HiCache transfers (pick up #42454 test…
metamergebot Oct 4, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
23 changes: 15 additions & 8 deletions python/sglang/srt/arg_groups/kv_cache_hook.py
Original file line number Diff line number Diff line change
Expand Up @@ -530,18 +530,10 @@ def handle_unified_memory_pool(server_args: Any) -> None:
)
if cfg.disaggregation_decode_retraction_backup == "host_pool":
model_config = model_config_of(server_args)
assert not cfg.disaggregation_decode_enable_radix_cache, (
"--enable-unified-memory host-pool decode retraction does not "
"support decode radix-cache H2D/D2H transfers yet."
)
assert mambaish_config(model_config) is None, (
"--enable-unified-memory host-pool decode retraction does not "
"support hybrid-Mamba models."
)
assert not model_config.is_hybrid_swa, (
"--enable-unified-memory host-pool decode retraction does not "
"support hybrid-SWA H2D/D2H transfers yet."
)
assert cfg.speculative_algorithm in (None, "DSPARK"), (
"--enable-unified-memory only supports --speculative-algorithm "
"DSPARK (chain draft); other speculative algorithms are not yet "
Expand Down Expand Up @@ -578,6 +570,21 @@ def handle_unified_memory_pool(server_args: Any) -> None:
"the LMCache offload path indexes the device buffers with the ids it "
"is handed, and under the unified pool those are VIRTUAL."
)
assert not cfg.enable_unified_cache_external_linker, (
"--enable-unified-memory does not support "
"--enable-unified-cache-external-linker: direct L3 transfers do not "
"preserve unified page-envelope indices and compaction lifetimes. "
"Use --enable-hierarchical-cache for supported L2/L3 transfers."
)
if cfg.enable_hierarchical_cache:
assert cfg.pp_size == 1, (
"--enable-unified-memory with hierarchical cache does not support "
"pipeline parallelism (--pp-size > 1)."
)
assert not envs.SGLANG_DISABLE_LAZY_COMPACTION.get(), (
"--enable-unified-memory with hierarchical cache requires lazy "
"compaction so pending H2D physical reservations remain stable."
)
Comment on lines +579 to +587

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

On main, handle_unified_memory_pool requires pp_size == 1 and lazy compaction only under PD disaggregation. Here both become startup assertions for any unified-memory + HiCache run. --enable-unified-memory --enable-hierarchical-cache --pp-size 2 now fails to launch, and so does the same command with SGLANG_DISABLE_LAZY_COMPACTION=1, the A/B and rollback escape hatch. Main starts both.

The lazy-compaction assert comes from the new unbound alloc_physical reservations. Main's bind-at-allocation path has no equivalent guard. The PP assert has no stated reason, and the description says the PP restrictions were removed.

Could you either keep main's binding path for these configurations, or give the reason and list both as trade-offs in the description?

Copy link
Copy Markdown
Collaborator Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Agreed. #42708 removes both asserts for unified memory with HiCache; the PD disaggregation asserts stay. test/registered/unit/server_args/test_unified_hicache_startup_args.py checks that --pp-size 2 and SGLANG_DISABLE_LAZY_COMPACTION=1 pass the argument layer. That is only an argument-layer check: pipeline-parallel serving with unified memory and HiCache has not been validated.

One correction on the lazy-compaction case. Main before #39479 also stopped at startup with SGLANG_DISABLE_LAZY_COMPACTION=1, just later:

  • The scheduler installs the host-transfer move gate for unified memory with HiCache (python/sglang/srt/managers/scheduler.py L598–L609 at 6cc661f, the parent of the Fix unified HiCache physical transfers #39479 merge).
  • install_move_gate asserts lazy compaction (python/sglang/srt/mem_cache/allocator/unified_sub_pool.py L242–L261, same commit).
  • SGLANG_DISABLE_LAZY_COMPACTION=1 turns lazy compaction off (python/sglang/srt/mem_cache/kv_cache_configurator.py L149–L152).

That allocator-level check predates #39479, and the follow-up leaves it unchanged.

if cfg.dcp_size > 1:
_validate_unified_memory_dcp(server_args)
# Prefill cuda-graph capture IS wired for the unified pool: the captured
Expand Down
29 changes: 12 additions & 17 deletions python/sglang/srt/disaggregation/decode.py
Original file line number Diff line number Diff line change
Expand Up @@ -787,20 +787,29 @@ def _match_prefix_and_lock(self, req: Req) -> DecodePrefixMatch:
Match a request against the decode-side radix cache, lock the matched
node to prevent eviction, and return the matched prefix information.
"""
max_prefix_len = None
if self._uses_swa_tail_prealloc():
fill_len = self._pre_alloc_fill_len(req)
max_prefix_len = fill_len - self._swa_tail_len(fill_len)
# Match and lock only reusable FULL KV. The entire SWA tail must be
# freshly allocated, including when the prefix comes from L2/L3.
result = match_prefix_for_req(
self.tree_cache,
req,
req.origin_input_ids,
cow_mamba=self.tree_cache.supports_mamba(),
include_req=True,
max_prefix_len=max_prefix_len,
)
# Keep aggregated scheduling semantics while preserving the SWA lock
# boundary needed for the matching dec_lock_ref; the full receipt
# travels on the req so every later release mirrors this acquire.
req.lock_receipt = self.tree_cache.inc_lock_ref(
result.last_device_node
).to_dec_params()
return self._build_decode_prefix_match(req, result)
return self._build_decode_prefix_match(
req, result, max_prefix_len=max_prefix_len
)

def _resolve_prefill_dp_rank(self, req: Req) -> Optional[int]:
prefill_info = self.kv_manager.prefill_info_table.get(_bootstrap_addr(req))
Expand Down Expand Up @@ -1422,20 +1431,6 @@ def pop_preallocated(

fill_len = self._pre_alloc_fill_len(decode_req.req)

# Cap full-attention prefix reuse at the sliding-window start so
# the SWA window lands entirely in the fresh delta, keeping
# alloc_extend_swa_tail's tail->full mapping in range. Costs reuse
# of only the last ~window_size full-attention tokens.
if uses_swa_tail_prealloc and prefix_len > 0:
swa_prefix_cap = fill_len - self._swa_tail_len(fill_len)
if prefix_len > swa_prefix_cap:
prefix_len = swa_prefix_cap
prefix_indices = prefix_indices[:prefix_len]
# Cap the prefill-committed prefix too: tokens past the
# cap are not device-resident, so prefill must transfer
# them.
total_prefix_len = prefix_len

# Decode transfers the SWA tail fresh, so retain only the
# full-attention prefix lock needed for reuse.
if (
Expand Down Expand Up @@ -2317,9 +2312,9 @@ def alloc_for_decode_prealloc(
# the live window tail.
kv_loc = allocator.alloc_extend_swa_tail(
prefix_lens=torch.tensor(
[prefix_len], dtype=torch.int64, device=device
[total_prefix_len], dtype=torch.int64, device=device
),
prefix_lens_cpu=torch.tensor([prefix_len], dtype=torch.int64),
prefix_lens_cpu=torch.tensor([total_prefix_len], dtype=torch.int64),
seq_lens=torch.tensor([fill_len], dtype=torch.int64, device=device),
seq_lens_cpu=torch.tensor([fill_len], dtype=torch.int64),
last_loc=last_loc,
Expand Down
7 changes: 5 additions & 2 deletions python/sglang/srt/disaggregation/decode_hicache_mixin.py
Original file line number Diff line number Diff line change
Expand Up @@ -61,7 +61,9 @@ class HiCacheRestoreResult(Enum):
class DecodeHiCachePreallocMixin:
"""HiCache hooks for ``DecodePreallocQueue``: issue prefetch + reserve tokens."""

def _build_decode_prefix_match(self, req: Req, result: Any) -> DecodePrefixMatch:
def _build_decode_prefix_match(
self, req: Req, result: Any, *, max_prefix_len: Optional[int] = None
) -> DecodePrefixMatch:
"""Convert a ``match_prefix_for_req`` result into ``DecodePrefixMatch``.

Performs the optional L3 storage hit length query when decode-side
Expand All @@ -79,7 +81,7 @@ def _build_decode_prefix_match(self, req: Req, result: Any) -> DecodePrefixMatch
last_host_node
):
matched_len = l1_prefix_len + l2_host_hit_length
suffix_tokens = req.origin_input_ids[matched_len:]
suffix_tokens = req.origin_input_ids[matched_len:max_prefix_len]
last_hash = self.tree_cache.get_last_hash_value(last_host_node)
prefix_keys = (
self.tree_cache.get_prefix_hash_values(last_host_node)
Expand Down Expand Up @@ -221,6 +223,7 @@ def _try_hicache_queue_load_back(self, dr: DecodeRequest) -> bool:
dr.req.origin_input_ids,
cow_mamba=False,
include_req=True,
max_prefix_len=pm.decode_prefix_len,
)
new_indices, restored_node = self.tree_cache.init_load_back(
InitLoadBackParams(
Expand Down
80 changes: 64 additions & 16 deletions python/sglang/srt/managers/cache_controller.py
Original file line number Diff line number Diff line change
Expand Up @@ -43,7 +43,7 @@
from sglang.srt.mem_cache.l2_transfer import L2Transfer, L2TransferEngine
from sglang.srt.mem_cache.memory_pool import MLATokenToKVPool
from sglang.srt.mem_cache.utils import get_storage_hash_str
from sglang.srt.runtime_context import get_parallel
from sglang.srt.runtime_context import get_memory, get_parallel
from sglang.srt.utils import get_device_module

logger = logging.getLogger(__name__)
Expand Down Expand Up @@ -842,6 +842,10 @@ def start_writing(self) -> None:
self._l2_transfers(host_indices, device_indices, pool_transfers)
)

self.mem_pool_device_allocator.set_hicache_transfer_done_event(
(id(self), "write"), completion.finish_event
)

self.ack_write_queue.append(
HiCacheAck(
start_event=completion.start_event,
Expand Down Expand Up @@ -986,6 +990,10 @@ def start_loading(self) -> int:
transfer_layer_id_max=self.transfer_layer_id_max,
)

self.mem_pool_device_allocator.set_hicache_transfer_done_event(
(id(self), "load"), completion.finish_event
)

self.ack_load_queue.append(
HiCacheAck(
start_event=completion.start_event,
Expand Down Expand Up @@ -1108,13 +1116,22 @@ def _page_transfer(self, operation: PrefetchOperation) -> int:
# Get one batch token, and update the completed_tokens if succeed
extra_info = HiCacheStorageExtraInfo(prefix_keys=prefix_keys)

hit_pages = self._page_transfer_kv_batch(
operation,
batch_hashes,
batch_host_indices,
extra_info,
kv_derived_transfers,
)
try:
hit_pages = self._page_transfer_kv_batch(
operation,
batch_hashes,
batch_host_indices,
extra_info,
kv_derived_transfers,
)
except Exception:
if not get_memory().enable_unified_memory:
raise
logger.exception(
"HiCache prefetch transfer failed for request %s",
operation.request_id,
)
hit_pages = 0
# Check termination
if hit_pages != len(batch_hashes):
all_success = False
Expand Down Expand Up @@ -1175,19 +1192,20 @@ def prefetch_io_aux_func(self):
while not self.storage_stop_event.is_set():
try:
operation = self.prefetch_buffer.get(block=True, timeout=1)
if operation is None:
continue
except Empty:
continue
if operation is None:
continue
try:
self._page_transfer(operation)

finally:
self.prefetch_sync_queue.put(
PrefetchAck(
rid=operation.request_id,
completed_req=True,
operation=operation,
)
)
except Empty:
continue

def prefetch_rate_limited(self) -> bool:
"""
Expand All @@ -1209,6 +1227,24 @@ def prefetch_rate_limited(self) -> bool:
# todo: more sophisticated rate limiting based on storage backend performance
return False

def alloc_prefetch_host_buffers(
self, operation: StorageOperation, need_size: int
) -> Optional[torch.Tensor]:
"""Allocate the host bounce for a storage hit."""
return self.mem_pool_host.alloc(need_size)

def can_fit_prefetch_host_buffers(
self, operation: StorageOperation, need_size: int
) -> bool:
"""Whether a prefetch bounce can fit when its host pools are empty."""
return need_size <= self.mem_pool_host.size

def free_prefetch_host_buffers(
self, operation: StorageOperation, host_indices: torch.Tensor
) -> None:
"""Roll back a hit-sized host bounce before transfer ownership moves."""
self.mem_pool_host.free(host_indices)

def _storage_hit_query(self, operation) -> tuple[list[str], int]:
last_hash = operation.last_hash
tokens_to_fetch = operation.token_ids
Expand Down Expand Up @@ -1243,10 +1279,22 @@ def prefetch_thread_func(self):
operation = self.prefetch_queue.get(block=True, timeout=1)
if operation is None:
continue
if operation.is_terminated():
try:
if operation.is_terminated():
hash_value, storage_hit_count = [], 0
else:
hash_value, storage_hit_count = self._storage_hit_query(
operation
)
except Exception:
if not get_memory().enable_unified_memory:
raise
logger.exception(
"HiCache storage query failed for request %s",
operation.request_id,
)
hash_value, storage_hit_count = [], 0
else:
hash_value, storage_hit_count = self._storage_hit_query(operation)

storage_hit_count_tensor = torch.tensor(
storage_hit_count, dtype=torch.int
)
Expand Down
Loading
Loading