Repository navigation
fix(grpc): drain in-flight requests on SIGTERM for graceful scale-down #1172
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Changes from all commits
f04cb75
d6aed74
db30688
263fd2c
ac0f8cb
90917f7
95ed234
File filter
Filter by extension
Conversations
Jump to
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change | ||||||||||||||||||||
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
|
|
@@ -21,7 +21,7 @@ | |||||||||||||||||||||
| from sglang.srt.disaggregation.utils import FAKE_BOOTSTRAP_HOST, DisaggregationMode | ||||||||||||||||||||||
| from sglang.srt.managers.disagg_service import start_disagg_service | ||||||||||||||||||||||
| from sglang.srt.server_args import ServerArgs | ||||||||||||||||||||||
| from sglang.srt.utils import kill_process_tree | ||||||||||||||||||||||
| from sglang.srt.utils import get_bool_env_var, kill_process_tree | ||||||||||||||||||||||
| from sglang.utils import get_exception_traceback | ||||||||||||||||||||||
| from smg_grpc_proto import sglang_scheduler_pb2, sglang_scheduler_pb2_grpc | ||||||||||||||||||||||
|
|
||||||||||||||||||||||
|
|
@@ -276,6 +276,10 @@ def _cert_config_fetcher(): | |||||||||||||||||||||
|
|
||||||||||||||||||||||
| def signal_handler(): | ||||||||||||||||||||||
| logger.info("Received shutdown signal") | ||||||||||||||||||||||
| # Flip health to NOT_SERVING and mark the request manager as | ||||||||||||||||||||||
| # draining so K8s stops routing new traffic to this pod and | ||||||||||||||||||||||
| # in-flight requests are allowed to finish. | ||||||||||||||||||||||
| servicer.begin_drain() | ||||||||||||||||||||||
|
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more.
Calling Useful? React with 👍 / 👎. |
||||||||||||||||||||||
| stop_event.set() | ||||||||||||||||||||||
|
|
||||||||||||||||||||||
| for sig in (signal.SIGTERM, signal.SIGINT): | ||||||||||||||||||||||
|
|
@@ -284,6 +288,53 @@ def signal_handler(): | |||||||||||||||||||||
| try: | ||||||||||||||||||||||
| await stop_event.wait() | ||||||||||||||||||||||
| finally: | ||||||||||||||||||||||
| # Drain phase: wait for in-flight requests to finish naturally. | ||||||||||||||||||||||
| # Mirrors tokenizer_manager.sigterm_watchdog. No in-process | ||||||||||||||||||||||
| # timeout; K8s terminationGracePeriodSeconds SIGKILL is the backstop. | ||||||||||||||||||||||
| # rid_to_state is only mutated from the event-loop thread, so a | ||||||||||||||||||||||
| # list() snapshot is safe without additional guarding. | ||||||||||||||||||||||
| while True: | ||||||||||||||||||||||
|
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more.
The new drain loop runs before the server is stopped, so the process still accepts Useful? React with 👍 / 👎. |
||||||||||||||||||||||
| if get_bool_env_var("SGL_FORCE_SHUTDOWN"): | ||||||||||||||||||||||
| logger.warning("SGL_FORCE_SHUTDOWN set; skipping drain") | ||||||||||||||||||||||
| break | ||||||||||||||||||||||
|
|
||||||||||||||||||||||
| # If handle_loop has died (e.g. fatal ZMQError, scheduler | ||||||||||||||||||||||
| # crash), scheduler outputs will never be forwarded and | ||||||||||||||||||||||
| # in-flight rids can never be marked finished — abort the | ||||||||||||||||||||||
| # drain instead of blocking until SIGKILL. | ||||||||||||||||||||||
| handle_loop_task = servicer.request_manager.handle_loop_task | ||||||||||||||||||||||
| if handle_loop_task is not None and handle_loop_task.done(): | ||||||||||||||||||||||
| logger.warning( | ||||||||||||||||||||||
| "handle_loop task has terminated; aborting drain and proceeding to shutdown" | ||||||||||||||||||||||
| ) | ||||||||||||||||||||||
| break | ||||||||||||||||||||||
|
|
||||||||||||||||||||||
| # Finished requests linger in rid_to_state for 5 s via the | ||||||||||||||||||||||
| # cleanup() task in _handle_batch_output; exclude them so | ||||||||||||||||||||||
| # the drain loop exits promptly once real work is done. | ||||||||||||||||||||||
| remaining_rids = [ | ||||||||||||||||||||||
| rid | ||||||||||||||||||||||
| for rid, state in servicer.request_manager.rid_to_state.items() | ||||||||||||||||||||||
| if not state.finished | ||||||||||||||||||||||
| ] | ||||||||||||||||||||||
|
Comment on lines
+315
to
+319
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more.
The drain loop decides shutdown readiness solely from Useful? React with 👍 / 👎. |
||||||||||||||||||||||
| remain_num_req = len(remaining_rids) | ||||||||||||||||||||||
| if remain_num_req == 0: | ||||||||||||||||||||||
| logger.info("Drain complete; no in-flight requests") | ||||||||||||||||||||||
| break | ||||||||||||||||||||||
|
|
||||||||||||||||||||||
| # Truncate rid list in logs: under high load there can be | ||||||||||||||||||||||
| # thousands of in-flight requests and the full list would | ||||||||||||||||||||||
| # flood the log every poll interval. | ||||||||||||||||||||||
| log_rids = remaining_rids[:10] | ||||||||||||||||||||||
| if remain_num_req > 10: | ||||||||||||||||||||||
| log_rids = log_rids + ["..."] | ||||||||||||||||||||||
| logger.info( | ||||||||||||||||||||||
| "Gracefully exiting... Remaining number of requests %d. Remaining requests %s", | ||||||||||||||||||||||
| remain_num_req, | ||||||||||||||||||||||
| log_rids, | ||||||||||||||||||||||
| ) | ||||||||||||||||||||||
|
Comment on lines
+331
to
+335
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. Logging the full list of
Suggested change
|
||||||||||||||||||||||
| await asyncio.sleep(5) | ||||||||||||||||||||||
|
Comment on lines
+296
to
+336
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. There is a potential hang in this drain loop if the background
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. A 5-second sleep interval for the drain loop is relatively long. It can delay the final termination of the pod by several seconds even after all in-flight requests have finished. Reducing this to 1 second would make the shutdown process more responsive without significantly increasing CPU overhead.
Suggested change
|
||||||||||||||||||||||
|
|
||||||||||||||||||||||
| logger.info("Shutting down gRPC server") | ||||||||||||||||||||||
|
|
||||||||||||||||||||||
| # Shutdown request manager first - this closes ZMQ sockets and stops background tasks | ||||||||||||||||||||||
|
|
||||||||||||||||||||||
| Original file line number | Diff line number | Diff line change |
|---|---|---|
|
|
@@ -212,6 +212,16 @@ async def Generate( | |
| """Handle generation requests with streaming responses.""" | ||
| logger.info(f"Receive generation request: {request.request_id}") | ||
|
|
||
| # Reject new RPCs once drain has started. K8s removes the pod from | ||
| # Service Endpoints when health flips to NOT_SERVING, but persistent | ||
| # clients or direct pod traffic could otherwise keep feeding work | ||
| # into rid_to_state and stall the drain loop indefinitely. | ||
| if self.request_manager.gracefully_exit: | ||
| await context.abort( | ||
| grpc.StatusCode.UNAVAILABLE, | ||
| "Server is shutting down", | ||
| ) | ||
|
Comment on lines
+219
to
+223
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more.
The new Useful? React with 👍 / 👎. |
||
|
|
||
| try: | ||
| # Convert gRPC request to internal format | ||
| tokenized_req = self._convert_generate_request(request) | ||
|
|
@@ -271,6 +281,13 @@ async def Embed( | |
| """Handle embedding requests.""" | ||
| logger.info(f"Receive embedding request: {request.request_id}") | ||
|
|
||
| # Reject new RPCs once drain has started (same rationale as Generate). | ||
| if self.request_manager.gracefully_exit: | ||
| await context.abort( | ||
| grpc.StatusCode.UNAVAILABLE, | ||
| "Server is shutting down", | ||
| ) | ||
|
|
||
| try: | ||
| tokenized_req = self._convert_embed_request(request) | ||
|
|
||
|
|
@@ -1098,6 +1115,17 @@ def _create_completion_response( | |
| ), | ||
| ) | ||
|
|
||
| def begin_drain(self) -> None: | ||
| """Mark the service as draining so health flips to NOT_SERVING. | ||
|
|
||
| Non-destructive: does not cancel in-flight requests. Safe to call | ||
| from a synchronous signal handler. Must be followed by shutdown() | ||
| (after the drain loop completes) for final cleanup. | ||
| """ | ||
| if self.health_servicer: | ||
| self.health_servicer.set_not_serving() | ||
| self.request_manager.begin_drain() | ||
|
|
||
| async def shutdown(self): | ||
| """Shutdown the service.""" | ||
| logger.info("Shutting down gRPC service") | ||
|
|
||
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
begin_drain()now setsgracefully_exitbefore real shutdown, but this branch still treats anyZMQErrorunder that flag as shutdown and exitshandle_loop. During a long drain window, a transient socket error would stop forwarding scheduler outputs to request queues, which can leave in-flight streams stuck and prevent the drain loop from ever completing cleanly. Drain and terminal-shutdown states should be separated for this error path.Useful? React with 👍 / 👎.