diff --git a/python/sglang/srt/managers/data_parallel_controller.py b/python/sglang/srt/managers/data_parallel_controller.py index 57632e146d94..ff7c15b805cd 100644 --- a/python/sglang/srt/managers/data_parallel_controller.py +++ b/python/sglang/srt/managers/data_parallel_controller.py @@ -559,6 +559,10 @@ def launch_tensor_parallel_group( self.max_total_num_tokens = scheduler_info[0]["max_total_num_tokens"] self.max_req_input_len = scheduler_info[0]["max_req_input_len"] + # Retain the full first child init dict so callers that build + # the upstream handshake (run_data_parallel_controller_process, + # below) can forward every field — not just the two limits. + self.scheduler_init_info = scheduler_info[0] def maybe_external_dp_rank_routing(self, req: Req): if req.routed_dp_rank is not None: @@ -649,10 +653,12 @@ def run_data_parallel_controller_process( proc.pid for proc in controller.scheduler_procs if proc is not None ] pipe_writer.send( + # Spread the first child scheduler's full init dict (status + + # both max_* limits + the multimodal/tokenizer metadata added + # by the get_init_info patch above) so dp_size > 1 doesn't + # strip them before they reach the gRPC servicer. { - "status": "ready", - "max_total_num_tokens": controller.max_total_num_tokens, - "max_req_input_len": controller.max_req_input_len, + **controller.scheduler_init_info, SCHEDULER_PIDS_ARG: scheduler_pids, } ) diff --git a/python/sglang/srt/managers/scheduler.py b/python/sglang/srt/managers/scheduler.py index 689cf9837bcc..6b9fdcc4cea6 100644 --- a/python/sglang/srt/managers/scheduler.py +++ b/python/sglang/srt/managers/scheduler.py @@ -1494,10 +1494,32 @@ def get_init_info(self) -> Dict[str, Any]: This method provides the initialization info needed by the tokenizer manager and other components to verify the scheduler is ready. """ + # Fields below `max_req_input_len` are read by smg-grpc-servicer's + # GetModelInfo bridge (sglang/server.py:81-97, servicer.py:325-329) + # out of scheduler_info via .get(..., default). Without them the + # gateway sees defaults — supports_vision in particular is always + # False, which makes multimodal workers invisible to image/audio + # routing. They are no-ops for HTTP-mode consumers. + hf_cfg = self.model_config.hf_config + pad_id = getattr(hf_cfg, "pad_token_id", None) + bos_id = getattr(hf_cfg, "bos_token_id", None) + # `hf_eos_token_id` is annotated Optional[Set[int]] but the + # producer accepts int/list/None inputs; coerce defensively. + eos_ids = self.model_config.hf_eos_token_id + eos_token_ids = sorted([eos_ids] if isinstance(eos_ids, int) else eos_ids or []) result_dict = { "status": "ready", "max_total_num_tokens": self.max_total_num_tokens, "max_req_input_len": self.max_req_input_len, + "is_generation": self.is_generation, + "supports_vision": self.model_config.is_multimodal, + "vocab_size": self.model_config.vocab_size, + "eos_token_ids": eos_token_ids, + # int32 proto field; coerce missing IDs to the smg-grpc-servicer + # int defaults (0 for pad, 1 for bos) so encoding doesn't crash + # on `None`. + "pad_token_id": pad_id if pad_id is not None else 0, + "bos_token_id": bos_id if bos_id is not None else 1, } return result_dict