diff --git a/.github/filters.yaml b/.github/filters.yaml index d66913fd2fc8..39b55cba9129 100644 --- a/.github/filters.yaml +++ b/.github/filters.yaml @@ -360,6 +360,7 @@ sidecar: - 'tests/utils/prometheus.py' - 'tests/utils/router_logs.py' - 'tests/utils/router_nvext.py' + - 'tests/utils/payloads.py' - 'Cargo.toml' - 'Cargo.lock' # Compliance policy and baseline changes can fail the sidecar image build. diff --git a/.github/workflows/shared-build-sidecar.yml b/.github/workflows/shared-build-sidecar.yml index 6219ad142563..75bd45a6099b 100644 --- a/.github/workflows/shared-build-sidecar.yml +++ b/.github/workflows/shared-build-sidecar.yml @@ -26,7 +26,7 @@ on: type: string default: '30' diff_event_context: - description: 'Compliance diff context (pr or push); empty disables baseline resolution' + description: 'Compliance diff context (pr, push, or nightly); empty disables baseline resolution' required: false type: string default: '' diff --git a/lib/sidecar/sglang/launch/disagg.sh b/lib/sidecar/sglang/launch/disagg.sh index 1937acdb111d..95ff980306fd 100755 --- a/lib/sidecar/sglang/launch/disagg.sh +++ b/lib/sidecar/sglang/launch/disagg.sh @@ -2,17 +2,16 @@ # SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 # -# Disaggregated serving through two SGLang native gRPC servers (2 GPUs). +# Disaggregated serving through two SGLang native gRPC servers (2 workers). # Requires an SGLang build with native gRPC sidecar and disaggregated serving support. set -e SCRIPT_DIR="$(dirname "$(readlink -f "$0")")" -export DYNAMO_HOME="${DYNAMO_HOME:-$(readlink -f "$SCRIPT_DIR/../../../..")}" # shellcheck disable=SC1091 # Resolved relative to this script at runtime. -source "$DYNAMO_HOME/examples/common/gpu_utils.sh" # build_sglang_gpu_mem_args +source "$SCRIPT_DIR/../../../../examples/common/gpu_utils.sh" # build_sglang_gpu_mem_args # shellcheck disable=SC1091 # Resolved relative to this script at runtime. -source "$DYNAMO_HOME/examples/common/launch_utils.sh" # print_launch_banner, wait_any_exit +source "$SCRIPT_DIR/../../../../examples/common/launch_utils.sh" # print_launch_banner, wait_any_exit MODEL="${MODEL:-Qwen/Qwen3-0.6B}" @@ -76,7 +75,7 @@ MAX_CONCURRENT_SEQS="${MAX_CONCURRENT_SEQS:-2}" HTTP_PORT="${DYN_HTTP_PORT:-8000}" GPU_MEM_ARGS=$(build_sglang_gpu_mem_args) -print_launch_banner "Launching SGLang Native-gRPC Sidecar (Disaggregated, 2 GPUs)" "$MODEL" "$HTTP_PORT" \ +print_launch_banner "Launching SGLang Native-gRPC Sidecar (Disaggregated, 2 workers)" "$MODEL" "$HTTP_PORT" \ "Prefill: GPU ${SGLANG_PREFILL_GPU}, HTTP http://${SGLANG_HOST}:${SGLANG_PREFILL_HTTP_PORT}, gRPC ${SGLANG_HOST}:${SGLANG_PREFILL_GRPC_PORT}" \ "Decode: GPU ${SGLANG_DECODE_GPU}, HTTP http://${SGLANG_HOST}:${SGLANG_DECODE_HTTP_PORT}, gRPC ${SGLANG_HOST}:${SGLANG_DECODE_GRPC_PORT}" \ "Bootstrap: ${SGLANG_BOOTSTRAP_HOST}:${SGLANG_DISAGGREGATION_BOOTSTRAP_PORT}" diff --git a/lib/sidecar/vllm/launch/disagg.sh b/lib/sidecar/vllm/launch/disagg.sh index 89f3c167fc89..7003d9d2139d 100755 --- a/lib/sidecar/vllm/launch/disagg.sh +++ b/lib/sidecar/vllm/launch/disagg.sh @@ -2,16 +2,15 @@ # SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 # -# Disaggregated serving through vLLM's native gRPC servers (2 GPUs). +# Disaggregated serving through vLLM's native gRPC servers (2 workers). set -e SCRIPT_DIR="$(dirname "$(readlink -f "$0")")" -export DYNAMO_HOME="${DYNAMO_HOME:-$(readlink -f "$SCRIPT_DIR/../../../..")}" # shellcheck disable=SC1091 # Resolved relative to this script at runtime. -source "$DYNAMO_HOME/examples/common/gpu_utils.sh" # build_vllm_gpu_mem_args +source "$SCRIPT_DIR/../../../../examples/common/gpu_utils.sh" # build_vllm_gpu_mem_args # shellcheck disable=SC1091 # Resolved relative to this script at runtime. -source "$DYNAMO_HOME/examples/common/launch_utils.sh" # print_launch_banner, wait_any_exit +source "$SCRIPT_DIR/../../../../examples/common/launch_utils.sh" # print_launch_banner, wait_any_exit MODEL="${MODEL:-Qwen/Qwen3-0.6B}" @@ -82,7 +81,7 @@ if [[ -z "$GPU_MEM_ARGS" ]]; then fi HTTP_PORT="${DYN_HTTP_PORT:-8000}" -print_launch_banner "Launching vLLM Native-gRPC Sidecar Disaggregated Serving (2 GPUs)" "$MODEL" "$HTTP_PORT" \ +print_launch_banner "Launching vLLM Native-gRPC Sidecar Disaggregated Serving (2 workers)" "$MODEL" "$HTTP_PORT" \ "Decode: GPU ${VLLM_DECODE_GPU}, gRPC 127.0.0.1:${VLLM_DECODE_GRPC_PORT}" \ "Prefill: GPU ${VLLM_PREFILL_GPU}, gRPC 127.0.0.1:${VLLM_PREFILL_GRPC_PORT}" diff --git a/tests/serve/test_sidecar.py b/tests/serve/test_sidecar.py index 727def2b28d7..c0be1f325929 100644 --- a/tests/serve/test_sidecar.py +++ b/tests/serve/test_sidecar.py @@ -1,7 +1,7 @@ # SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 -"""E2E coverage for lib/sidecar/{vllm,sglang,trtllm}/launch/agg.sh (native-gRPC sidecar + engine).""" +"""E2E coverage for native-gRPC sidecar launch scripts.""" import dataclasses import os @@ -18,8 +18,8 @@ from tests.utils.constants import DynamoPortRange from tests.utils.engine_process import EngineConfig from tests.utils.gpu_args import map_cuda_visible_devices -from tests.utils.payload_builder import chat_payload_default -from tests.utils.payloads import ChatPayload +from tests.utils.payload_builder import LONG_PROMPT_FOR_CACHING, chat_payload_default +from tests.utils.payloads import ChatPayload, DisaggregatedChatPayload from tests.utils.port_utils import reserved_ports vllm_sidecar_dir = os.environ.get("VLLM_SIDECAR_DIR") or os.path.join( @@ -40,6 +40,23 @@ def _sidecar_worker_gpu_env(backend: str) -> dict[str, str]: return {f"{backend.upper()}_WORKER{index + 1}_GPU": device for index in range(2)} +def _disaggregated_chat_payload() -> DisaggregatedChatPayload: + return DisaggregatedChatPayload( + body={ + "messages": [{"role": "user", "content": LONG_PROMPT_FOR_CACHING}], + "max_tokens": 64, + "n": 1, + "temperature": 0, + "stream": False, + "nvext": {"extra_fields": ["worker_id"]}, + }, + repeat_count=1, + expected_response=[], + expected_log=[], + expected_num_choices=1, + ) + + # Sequential stage only: no profiled_vram_gib mark yet, since actual peak VRAM # has not been profiled for the sidecar launch path. Add one once measured, to # admit these into the parallel stage alongside the equivalent dynamo.{backend} @@ -100,6 +117,46 @@ def _sidecar_worker_gpu_env(backend: str) -> dict[str, str]: chat_payload_default(), ], ), + # Prefill/decode handoff is a critical native-sidecar path. + "vllm_disaggregated": EngineConfig( + name="vllm_disaggregated", + directory=vllm_sidecar_dir, + script_name="disagg.sh", + marks=[ + pytest.mark.vllm, + pytest.mark.gpu_1, + pytest.mark.pre_merge, + pytest.mark.post_merge, + pytest.mark.nightly, + pytest.mark.timeout(1200), + pytest.mark.requested_vllm_kv_cache_bytes(1119388000), + ], + model="Qwen/Qwen3-0.6B", + health_check_workers=True, + health_check_worker_count=2, + env={"PYTHONUNBUFFERED": "1", "MAX_MODEL_LEN": "2048"}, + request_payloads=[_disaggregated_chat_payload()], + ), + "sglang_disaggregated": EngineConfig( + name="sglang_disaggregated", + directory=sglang_sidecar_dir, + script_name="disagg.sh", + script_args=["--disable-cuda-graph"], + marks=[ + pytest.mark.sglang, + pytest.mark.gpu_1, + pytest.mark.pre_merge, + pytest.mark.post_merge, + pytest.mark.nightly, + pytest.mark.timeout(1200), + pytest.mark.requested_sglang_kv_tokens(2048), + ], + model="Qwen/Qwen3-0.6B", + health_check_workers=True, + health_check_worker_count=2, + env={"PYTHONUNBUFFERED": "1", "MAX_MODEL_LEN": "2048"}, + request_payloads=[_disaggregated_chat_payload()], + ), } @@ -120,19 +177,51 @@ def test_serve_deployment( dynamo_dynamic_ports, num_system_ports, predownload_models, + monkeypatch, ): - """ - Launch a lib/sidecar//launch/agg.sh script end-to-end (Dynamo - frontend + native-gRPC engine + dynamo--sidecar) and confirm it - serves a real chat completion. - """ + """Launch a native engine and sidecar deployment and validate chat completion.""" assert ( num_system_ports >= 2 ), "serve tests require at least SYSTEM_PORT1 + SYSTEM_PORT2" config = dataclasses.replace( sidecar_config_test, frontend_port=dynamo_dynamic_ports.frontend_port ) - if config.name == "vllm_aggregated": + if config.name.endswith("_disaggregated"): + monkeypatch.delenv("DYN_NAMESPACE_WORKER_SUFFIX", raising=False) + monkeypatch.setenv("DYN_REQUEST_PLANE", "tcp") + backend = config.name.removesuffix("_disaggregated") + roles = ("DECODE", "PREFILL") if backend == "vllm" else ("PREFILL", "DECODE") + device = map_cuda_visible_devices([0], os.environ.get("CUDA_VISIBLE_DEVICES")) + assert device != "-1", "One visible GPU is required" + engine_env = { + **{f"{backend.upper()}_{role}_GPU": device for role in roles}, + "DYN_NAMESPACE": f"sidecar-disagg-{generate_random_suffix()}", + "MODEL": config.model, + } + num_engine_ports = {"vllm": 4, "sglang": 5}[backend] + with reserved_ports( + num_engine_ports, start_port=DynamoPortRange.SERVE.value + ) as engine_ports: + for index, role in enumerate(roles): + prefix = f"{backend.upper()}_{role}" + engine_env[f"{prefix}_HTTP_PORT"] = str(engine_ports[index * 2]) + engine_env[f"{prefix}_GRPC_PORT"] = str(engine_ports[index * 2 + 1]) + if backend == "vllm": + engine_env[f"{prefix}_NIXL_SIDE_CHANNEL_PORT"] = str( + dynamo_dynamic_ports.nixl_side_channel_ports[index] + ) + if backend == "vllm": + engine_env["VLLM_PREFILL_KV_EVENT_PORT"] = str( + dynamo_dynamic_ports.kv_event_ports[1] + ) + elif backend == "sglang": + engine_env["SGLANG_DISAGGREGATION_BOOTSTRAP_PORT"] = str( + engine_ports[4] + ) + run_serve_deployment( + config, request, ports=dynamo_dynamic_ports, extra_env=engine_env + ) + elif config.name == "vllm_aggregated": with reserved_ports(2, start_port=DynamoPortRange.SERVE.value) as engine_ports: run_serve_deployment( config, diff --git a/tests/utils/payloads.py b/tests/utils/payloads.py index 363bb1e63835..5a8ed45654ba 100644 --- a/tests/utils/payloads.py +++ b/tests/utils/payloads.py @@ -33,7 +33,11 @@ from tests.utils.http_checks import check_health_generate as check_health_generate from tests.utils.http_checks import check_models_api as check_models_api from tests.utils.prometheus import find_metric_samples, sum_metric_samples -from tests.utils.router_nvext import RouterNvextExpectation, validate_router_nvext +from tests.utils.router_nvext import ( + RouterNvextExpectation, + require_router_worker_id, + validate_router_nvext, +) logger = logging.getLogger(__name__) @@ -215,6 +219,42 @@ def validate(self, response: Any, content: str) -> None: ) +class DisaggregatedChatPayload(ChatPayload): + """Require a completed chat request served by distinct prefill and decode workers.""" + + def validate(self, response: Any, content: str) -> None: + super().validate(response, content) + result = response.json() + choices = result["choices"] + if len(choices) != 1: + raise AssertionError(f"Expected one completion, got {choices!r}") + if not isinstance(content, str) or not content.strip(): + raise AssertionError("Completion is empty") + if choices[0].get("finish_reason") not in {"stop", "length"}: + raise AssertionError(f"Unexpected finish reason: {choices[0]!r}") + + usage = result.get("usage") + if not isinstance(usage, dict): + raise AssertionError(f"Missing usage: {result!r}") + prompt_tokens = usage.get("prompt_tokens") + completion_tokens = usage.get("completion_tokens") + if type(prompt_tokens) is not int or prompt_tokens <= 0: + raise AssertionError(f"Expected positive prompt usage: {usage!r}") + if type(completion_tokens) is not int or completion_tokens <= 1: + raise AssertionError( + f"Expected decode to generate more than the prefill token: {usage!r}" + ) + + workers = require_router_worker_id(result, context=type(self).__name__) + for role in ("prefill_worker_id", "decode_worker_id"): + if type(workers.get(role)) is not int or workers[role] < 0: + raise AssertionError(f"Expected a valid {role}: {dict(workers)!r}") + if workers["prefill_worker_id"] == workers["decode_worker_id"]: + raise AssertionError( + f"Expected distinct prefill and decode workers: {dict(workers)!r}" + ) + + class RouterNvextChatPayload(ChatPayload): """Chat payload that validates structured router metadata in nvext."""