Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 1 addition & 7 deletions tests/v1/core/test_scheduler.py
Original file line number Diff line number Diff line change
Expand Up @@ -1562,13 +1562,7 @@ def test_scheduler_reset_prefix_cache():
assert not scheduler.reset_prefix_cache()
scheduler.aux_output_connector.reset.assert_not_called()

with pytest.raises(RuntimeError, match=r"pause\(mode='keep'\)"):
scheduler.reset_prefix_cache(reset_running_requests=True)

# pause(mode="keep") also waits for scheduled model outputs to drain.
scheduler.set_pause_state(PauseState.PAUSED_ALL)
with pytest.raises(RuntimeError, match="model output is in flight"):
scheduler.reset_prefix_cache(reset_running_requests=True)
# Pause completes pending model outputs before the caller resets the scheduler.
for request in requests:
request.num_in_flight_tokens = 0

Expand Down
10 changes: 0 additions & 10 deletions vllm/v1/core/sched/scheduler.py
Original file line number Diff line number Diff line change
Expand Up @@ -2781,16 +2781,6 @@ def reset_prefix_cache(
Otherwise, this method will only reset the KV prefix cache when there
is no running requests taking KV cache.
"""
if reset_running_requests and self.aux_output_connector is not None:
if self._pause_state != PauseState.PAUSED_ALL:
raise RuntimeError(
"AuxOutput Connector only supports resetting running requests "
"after pause(mode='keep')."
)
if any(request.num_in_flight_tokens for request in self.requests.values()):
raise RuntimeError(
"AuxOutput Connector cannot reset while model output is in flight."
)
if reset_running_requests:
# For logging.
timestamp = time.monotonic()
Expand Down
Loading