From 0ed598754ba359742279b041227c23becf949e7b Mon Sep 17 00:00:00 2001 From: alphabetc1 <2508695655@qq.com> Date: Wed, 26 Aug 2026 09:50:30 +0800 Subject: [PATCH 1/3] fix: unified radix cache tests write markers into the pool, not a copy Co-Authored-By: Claude Opus 5 (1M context) --- .../mem_cache/test_unified_radix_cache_unittest.py | 14 ++++++++------ 1 file changed, 8 insertions(+), 6 deletions(-) diff --git a/test/registered/unit/mem_cache/test_unified_radix_cache_unittest.py b/test/registered/unit/mem_cache/test_unified_radix_cache_unittest.py index 67a8a94a0e21..7ae1b0c8c5c7 100644 --- a/test/registered/unit/mem_cache/test_unified_radix_cache_unittest.py +++ b/test/registered/unit/mem_cache/test_unified_radix_cache_unittest.py @@ -3963,10 +3963,11 @@ def _get_full_kv_pool(self, allocator): def _fill_full_kv(self, allocator, indices, marker): kv_pool = self._get_full_kv_pool(allocator) layer_id = kv_pool.start_layer - k_buf = kv_pool.get_key_buffer(layer_id) - v_buf = kv_pool.get_value_buffer(layer_id) - k_buf[indices].fill_(marker) - v_buf[indices].fill_(marker + 1) + # Assign through __setitem__: `buf[indices]` is an advanced-index copy, + # so `.fill_()` on it leaves the pool untouched and every bytes-match + # assertion downstream compares zeros to zeros. + kv_pool.get_key_buffer(layer_id)[indices] = marker + kv_pool.get_value_buffer(layer_id)[indices] = marker + 1 def _snapshot_full_kv(self, allocator, indices): kv_pool = self._get_full_kv_pool(allocator) @@ -3981,9 +3982,10 @@ def _fill_mamba_state(self, req_to_token_pool, indices, marker): return mamba_indices = indices.reshape(-1) mamba_cache = req_to_token_pool.mamba_pool.mamba_cache - mamba_cache.temporal[:, mamba_indices].fill_(marker) + # See _fill_full_kv: assign, never fill_ an advanced-index copy. + mamba_cache.temporal[:, mamba_indices] = marker for offset, conv_buf in enumerate(mamba_cache.conv, start=1): - conv_buf[:, mamba_indices].fill_(marker + offset) + conv_buf[:, mamba_indices] = marker + offset def _snapshot_mamba_state(self, req_to_token_pool, indices): mamba_indices = indices.reshape(-1) From c12432c8b9dc19eb60c0a94edd7f9687959b0021 Mon Sep 17 00:00:00 2001 From: alphabetc1 <2508695655@qq.com> Date: Wed, 26 Aug 2026 09:51:06 +0800 Subject: [PATCH 2/3] fix: charge the Mamba slot a prefill admission reserved but never debited Co-Authored-By: Claude Opus 5 (1M context) --- python/sglang/srt/managers/schedule_policy.py | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/python/sglang/srt/managers/schedule_policy.py b/python/sglang/srt/managers/schedule_policy.py index 34472fce20c5..74990e7ca835 100644 --- a/python/sglang/srt/managers/schedule_policy.py +++ b/python/sglang/srt/managers/schedule_policy.py @@ -1233,7 +1233,11 @@ def add_one_req( total_tokens = cand_extend_input_len + max_new + self.page_size # Shared Mamba pool: fold the new mamba state's shared-gap cost into # `total_tokens` so both `rem_total_tokens` gates reflect the joint budget. - total_tokens += self._mamba_gap_budget_for_req(req) + # Read once, before init_load_back below binds the request's slot: a + # second read afterwards returns 0, and the debit would then miss the + # slot this gate just reserved. + mamba_gap_reserve = self._mamba_gap_budget_for_req(req) + total_tokens += mamba_gap_reserve # adjusting the input_tokens based on host_hit_length and page_size real_input_tokens = cand_extend_input_len - req.host_hit_length @@ -1379,7 +1383,7 @@ def add_one_req( CLIP_MAX_NEW_TOKENS, ), req.retracted_stain, - mamba_gap_reserve=self._mamba_gap_budget_for_req(req), + mamba_gap_reserve=mamba_gap_reserve, host_hit_len=req.host_hit_length, storage_hit_len=req.storage_hit_length, ) @@ -1427,7 +1431,7 @@ def add_one_req( trunc_len, 0, req.retracted_stain, - mamba_gap_reserve=self._mamba_gap_budget_for_req(req), + mamba_gap_reserve=mamba_gap_reserve, host_hit_len=req.host_hit_length, storage_hit_len=req.storage_hit_length, ) From c0156fe482e8e358591a225e3b75d9148ff2584a Mon Sep 17 00:00:00 2001 From: alphabetc1 <2508695655@qq.com> Date: Wed, 2 Sep 2026 01:56:04 +0800 Subject: [PATCH 3/3] refactor: trim the comments on the mamba gap reserve and marker writes Co-Authored-By: Claude Opus 5 (1M context) --- python/sglang/srt/managers/schedule_policy.py | 5 ++--- .../unit/mem_cache/test_unified_radix_cache_unittest.py | 5 +---- 2 files changed, 3 insertions(+), 7 deletions(-) diff --git a/python/sglang/srt/managers/schedule_policy.py b/python/sglang/srt/managers/schedule_policy.py index 461cfaa945a1..4adee0b93e58 100644 --- a/python/sglang/srt/managers/schedule_policy.py +++ b/python/sglang/srt/managers/schedule_policy.py @@ -1202,9 +1202,8 @@ def add_one_req( total_tokens = cand_extend_input_len + max_new + self.page_size # Shared Mamba pool: fold the new mamba state's shared-gap cost into # `total_tokens` so both `rem_total_tokens` gates reflect the joint budget. - # Read once, before init_load_back below binds the request's slot: a - # second read afterwards returns 0, and the debit would then miss the - # slot this gate just reserved. + # Read before `init_load_back` binds `req.mamba_pool_idx` — after that + # this returns 0, so the debit sites below reuse the value. mamba_gap_reserve = self._mamba_gap_budget_for_req(req) total_tokens += mamba_gap_reserve diff --git a/test/registered/unit/mem_cache/test_unified_radix_cache_unittest.py b/test/registered/unit/mem_cache/test_unified_radix_cache_unittest.py index 85314aa98d77..6fc4c1689145 100644 --- a/test/registered/unit/mem_cache/test_unified_radix_cache_unittest.py +++ b/test/registered/unit/mem_cache/test_unified_radix_cache_unittest.py @@ -4955,9 +4955,7 @@ def _get_full_kv_pool(self, allocator): def _fill_full_kv(self, allocator, indices, marker): kv_pool = self._get_full_kv_pool(allocator) layer_id = kv_pool.start_layer - # Assign through __setitem__: `buf[indices]` is an advanced-index copy, - # so `.fill_()` on it leaves the pool untouched and every bytes-match - # assertion downstream compares zeros to zeros. + # `buf[indices]` is an advanced-index copy — assign, never `fill_()`. kv_pool.get_key_buffer(layer_id)[indices] = marker kv_pool.get_value_buffer(layer_id)[indices] = marker + 1 @@ -4974,7 +4972,6 @@ def _fill_mamba_state(self, req_to_token_pool, indices, marker): return mamba_indices = indices.reshape(-1) mamba_cache = req_to_token_pool.mamba_pool.mamba_cache - # See _fill_full_kv: assign, never fill_ an advanced-index copy. mamba_cache.temporal[:, mamba_indices] = marker for offset, conv_buf in enumerate(mamba_cache.conv, start=1): conv_buf[:, mamba_indices] = marker + offset