diff --git a/tests/v1/core/test_scheduler.py b/tests/v1/core/test_scheduler.py index 91c93ed68fda..d439a197e926 100644 --- a/tests/v1/core/test_scheduler.py +++ b/tests/v1/core/test_scheduler.py @@ -1562,13 +1562,7 @@ def test_scheduler_reset_prefix_cache(): assert not scheduler.reset_prefix_cache() scheduler.aux_output_connector.reset.assert_not_called() - with pytest.raises(RuntimeError, match=r"pause\(mode='keep'\)"): - scheduler.reset_prefix_cache(reset_running_requests=True) - - # pause(mode="keep") also waits for scheduled model outputs to drain. - scheduler.set_pause_state(PauseState.PAUSED_ALL) - with pytest.raises(RuntimeError, match="model output is in flight"): - scheduler.reset_prefix_cache(reset_running_requests=True) + # Pause completes pending model outputs before the caller resets the scheduler. for request in requests: request.num_in_flight_tokens = 0 diff --git a/vllm/v1/core/sched/scheduler.py b/vllm/v1/core/sched/scheduler.py index e4af90715443..24ecc4534d6d 100644 --- a/vllm/v1/core/sched/scheduler.py +++ b/vllm/v1/core/sched/scheduler.py @@ -2781,16 +2781,6 @@ def reset_prefix_cache( Otherwise, this method will only reset the KV prefix cache when there is no running requests taking KV cache. """ - if reset_running_requests and self.aux_output_connector is not None: - if self._pause_state != PauseState.PAUSED_ALL: - raise RuntimeError( - "AuxOutput Connector only supports resetting running requests " - "after pause(mode='keep')." - ) - if any(request.num_in_flight_tokens for request in self.requests.values()): - raise RuntimeError( - "AuxOutput Connector cannot reset while model output is in flight." - ) if reset_running_requests: # For logging. timestamp = time.monotonic()