From 4cc073cf2bd7e57209a0eaa8058299d4ead3db61 Mon Sep 17 00:00:00 2001 From: Roger Wang Date: Mon, 10 Aug 2026 21:39:31 -0700 Subject: [PATCH] [Kimi K3] Don't cut prefill at the partial-tail hash boundary mid-block In "align" mamba cache mode `_mamba_block_aligned_split` adds one extra stop at the prompt's last hash boundary so the partial-tail entry can be registered. That boundary is off the block grid by construction whenever `hash_block_size < block_size`, which is fine when the chunk *starts* on the grid -- the state at the start was materialized and the tail write is the only unaligned one. It is not fine when the request resumes mid-block. Then the block boundary already crossed was never materialized, and the extra stop cuts the chunk at a second unaligned position instead of re-aligning, leaving the SSM state wrong for the rest of the prefill. The production config that hit this is Kimi-K3: block_size 1536, `--prefix-match-unit 128`, external prefix hit at 24960 = 16*1536 + 384, prompt 25297. The re-align stop 26112 is past `last_cache_position` (24576) so it is suppressed; 24576 is behind the resume point; the tail boundary 25216 is left as the only stop and prefill is cut 640 tokens off the grid. Observed end-to-end as a silent HTTP 200 with `finish_reason: stop`, empty content, and the model resuming from the middle of a structural marker it never opened. Require a block-aligned start for that stop. Aligned starts keep registering the partial tail, so the existing behaviour and its test are unchanged; a mid-block resume now runs the fresh tail in one chunk, which is what the measured-clean configurations already did. Signed-off-by: Roger Wang Co-Authored-By: Claude Opus 5 (1M context) Signed-off-by: Yifan Qiao --- .../test_partial_prefix_cache_hits.py | 43 +++++++++++++++++++ vllm/v1/core/sched/scheduler.py | 9 ++-- 2 files changed, 49 insertions(+), 3 deletions(-) diff --git a/tests/v1/core/prefix_cache/test_partial_prefix_cache_hits.py b/tests/v1/core/prefix_cache/test_partial_prefix_cache_hits.py index 5f549ec8df7e..258f0a32460f 100644 --- a/tests/v1/core/prefix_cache/test_partial_prefix_cache_hits.py +++ b/tests/v1/core/prefix_cache/test_partial_prefix_cache_hits.py @@ -113,6 +113,49 @@ def test_mamba_align_split_partial_tail_schedule(): assert split(self=mock, request=req2, num_new_tokens=1000) == 512 +def test_mamba_align_split_skips_partial_tail_when_resumed_mid_block(): + """A request resuming mid-block must not stop at the partial-tail hash + boundary: that boundary is off the block grid, so the chunk would end with + the SSM state at a second unaligned position instead of re-aligning. + + Production config that hit this (Kimi-K3): block=1536, hash=128, + prompt=25297, external prefix hit at 24960 (= 16*1536 + 384, off grid). + The block-boundary stop 26112 is past `last_cache_position` (24576) so it + is suppressed, 24576 is behind the resume point, and the tail boundary + 25216 was left as the only stop -- cutting prefill 640 tokens off the grid. + """ + block_size = 1536 + hash_block_size = 128 + mock = SimpleNamespace( + cache_config=SimpleNamespace(block_size=block_size), + max_num_scheduled_tokens=32768, + scheduler_config=SimpleNamespace(long_prefill_token_threshold=8192), + use_eagle=False, + hash_block_size=hash_block_size, + dcp_world_size=1, + scheduler_block_size=block_size, + mamba_partial_cache_hit=True, + ) + split = Scheduler._mamba_block_aligned_split + + req = make_request("0", [0] * 25297, hash_block_size, sha256) + # Resumed off the block grid: the fresh tail runs in one chunk. + req.num_computed_tokens = 0 + assert ( + split( + self=mock, + request=req, + num_new_tokens=337, + num_external_computed_tokens=24960, + ) + == 337 + ) + + # A block-aligned start still registers the partial tail (24576 -> 25216). + req.num_computed_tokens = 24576 + assert split(self=mock, request=req, num_new_tokens=721) == 640 + + def test_mamba_align_split_when_block_exceeds_scheduling_budget(): """Sub-block chunks make progress only when no step can fit a full block.""" block_size = 11392 diff --git a/vllm/v1/core/sched/scheduler.py b/vllm/v1/core/sched/scheduler.py index a781b4580149..664b9837d1a5 100644 --- a/vllm/v1/core/sched/scheduler.py +++ b/vllm/v1/core/sched/scheduler.py @@ -407,10 +407,13 @@ def _mamba_block_aligned_split( next_block_boundary if start % block_size != 0 else 0, # Never run past the last cacheable block boundary mid-chunk. last_cache_position, - # Fine-grained hits: the prompt's partial-tail entry can only be - # registered by a chunk ending exactly at its last hash boundary. + # Register the partial tail only from an aligned start; otherwise + # this stop would strand the SSM state off the block grid. tail_boundary - if last_cache_position < tail_boundary < request.num_prompt_tokens + if ( + start % block_size == 0 + and last_cache_position < tail_boundary < request.num_prompt_tokens + ) else 0, # Marconi shared-prefix junction, block-floored (a sub-block # junction's state is not separately cacheable): cache its state