Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions .github/filters.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -360,6 +360,7 @@ sidecar:
- 'tests/utils/prometheus.py'
- 'tests/utils/router_logs.py'
- 'tests/utils/router_nvext.py'
- 'tests/utils/payloads.py'
- 'Cargo.toml'
- 'Cargo.lock'
# Compliance policy and baseline changes can fail the sidecar image build.
Expand Down
2 changes: 1 addition & 1 deletion .github/workflows/shared-build-sidecar.yml
Original file line number Diff line number Diff line change
Expand Up @@ -26,7 +26,7 @@ on:
type: string
default: '30'
diff_event_context:
description: 'Compliance diff context (pr or push); empty disables baseline resolution'
description: 'Compliance diff context (pr, push, or nightly); empty disables baseline resolution'
required: false
type: string
default: ''
Expand Down
9 changes: 4 additions & 5 deletions lib/sidecar/sglang/launch/disagg.sh
Original file line number Diff line number Diff line change
Expand Up @@ -2,17 +2,16 @@
# SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0
#
# Disaggregated serving through two SGLang native gRPC servers (2 GPUs).
# Disaggregated serving through two SGLang native gRPC servers (2 workers).
# Requires an SGLang build with native gRPC sidecar and disaggregated serving support.

set -e

SCRIPT_DIR="$(dirname "$(readlink -f "$0")")"
export DYNAMO_HOME="${DYNAMO_HOME:-$(readlink -f "$SCRIPT_DIR/../../../..")}"
# shellcheck disable=SC1091 # Resolved relative to this script at runtime.
source "$DYNAMO_HOME/examples/common/gpu_utils.sh" # build_sglang_gpu_mem_args
source "$SCRIPT_DIR/../../../../examples/common/gpu_utils.sh" # build_sglang_gpu_mem_args
# shellcheck disable=SC1091 # Resolved relative to this script at runtime.
source "$DYNAMO_HOME/examples/common/launch_utils.sh" # print_launch_banner, wait_any_exit
source "$SCRIPT_DIR/../../../../examples/common/launch_utils.sh" # print_launch_banner, wait_any_exit

MODEL="${MODEL:-Qwen/Qwen3-0.6B}"

Expand Down Expand Up @@ -76,7 +75,7 @@ MAX_CONCURRENT_SEQS="${MAX_CONCURRENT_SEQS:-2}"
HTTP_PORT="${DYN_HTTP_PORT:-8000}"
GPU_MEM_ARGS=$(build_sglang_gpu_mem_args)

print_launch_banner "Launching SGLang Native-gRPC Sidecar (Disaggregated, 2 GPUs)" "$MODEL" "$HTTP_PORT" \
print_launch_banner "Launching SGLang Native-gRPC Sidecar (Disaggregated, 2 workers)" "$MODEL" "$HTTP_PORT" \
"Prefill: GPU ${SGLANG_PREFILL_GPU}, HTTP http://${SGLANG_HOST}:${SGLANG_PREFILL_HTTP_PORT}, gRPC ${SGLANG_HOST}:${SGLANG_PREFILL_GRPC_PORT}" \
"Decode: GPU ${SGLANG_DECODE_GPU}, HTTP http://${SGLANG_HOST}:${SGLANG_DECODE_HTTP_PORT}, gRPC ${SGLANG_HOST}:${SGLANG_DECODE_GRPC_PORT}" \
"Bootstrap: ${SGLANG_BOOTSTRAP_HOST}:${SGLANG_DISAGGREGATION_BOOTSTRAP_PORT}"
Expand Down
9 changes: 4 additions & 5 deletions lib/sidecar/vllm/launch/disagg.sh
Original file line number Diff line number Diff line change
Expand Up @@ -2,16 +2,15 @@
# SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0
#
# Disaggregated serving through vLLM's native gRPC servers (2 GPUs).
# Disaggregated serving through vLLM's native gRPC servers (2 workers).

set -e

SCRIPT_DIR="$(dirname "$(readlink -f "$0")")"
export DYNAMO_HOME="${DYNAMO_HOME:-$(readlink -f "$SCRIPT_DIR/../../../..")}"
# shellcheck disable=SC1091 # Resolved relative to this script at runtime.
source "$DYNAMO_HOME/examples/common/gpu_utils.sh" # build_vllm_gpu_mem_args
source "$SCRIPT_DIR/../../../../examples/common/gpu_utils.sh" # build_vllm_gpu_mem_args
# shellcheck disable=SC1091 # Resolved relative to this script at runtime.
source "$DYNAMO_HOME/examples/common/launch_utils.sh" # print_launch_banner, wait_any_exit
source "$SCRIPT_DIR/../../../../examples/common/launch_utils.sh" # print_launch_banner, wait_any_exit

MODEL="${MODEL:-Qwen/Qwen3-0.6B}"

Expand Down Expand Up @@ -82,7 +81,7 @@ if [[ -z "$GPU_MEM_ARGS" ]]; then
fi

HTTP_PORT="${DYN_HTTP_PORT:-8000}"
print_launch_banner "Launching vLLM Native-gRPC Sidecar Disaggregated Serving (2 GPUs)" "$MODEL" "$HTTP_PORT" \
print_launch_banner "Launching vLLM Native-gRPC Sidecar Disaggregated Serving (2 workers)" "$MODEL" "$HTTP_PORT" \
"Decode: GPU ${VLLM_DECODE_GPU}, gRPC 127.0.0.1:${VLLM_DECODE_GRPC_PORT}" \
"Prefill: GPU ${VLLM_PREFILL_GPU}, gRPC 127.0.0.1:${VLLM_PREFILL_GRPC_PORT}"

Expand Down
107 changes: 98 additions & 9 deletions tests/serve/test_sidecar.py
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
# SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0

"""E2E coverage for lib/sidecar/{vllm,sglang,trtllm}/launch/agg.sh (native-gRPC sidecar + engine)."""
"""E2E coverage for native-gRPC sidecar launch scripts."""

import dataclasses
import os
Expand All @@ -18,8 +18,8 @@
from tests.utils.constants import DynamoPortRange
from tests.utils.engine_process import EngineConfig
from tests.utils.gpu_args import map_cuda_visible_devices
from tests.utils.payload_builder import chat_payload_default
from tests.utils.payloads import ChatPayload
from tests.utils.payload_builder import LONG_PROMPT_FOR_CACHING, chat_payload_default
from tests.utils.payloads import ChatPayload, DisaggregatedChatPayload
from tests.utils.port_utils import reserved_ports

vllm_sidecar_dir = os.environ.get("VLLM_SIDECAR_DIR") or os.path.join(
Expand All @@ -40,6 +40,23 @@ def _sidecar_worker_gpu_env(backend: str) -> dict[str, str]:
return {f"{backend.upper()}_WORKER{index + 1}_GPU": device for index in range(2)}


def _disaggregated_chat_payload() -> DisaggregatedChatPayload:
return DisaggregatedChatPayload(
body={
"messages": [{"role": "user", "content": LONG_PROMPT_FOR_CACHING}],
"max_tokens": 64,
"n": 1,
"temperature": 0,
"stream": False,
"nvext": {"extra_fields": ["worker_id"]},
},
repeat_count=1,
expected_response=[],
expected_log=[],
expected_num_choices=1,
)


# Sequential stage only: no profiled_vram_gib mark yet, since actual peak VRAM
# has not been profiled for the sidecar launch path. Add one once measured, to
# admit these into the parallel stage alongside the equivalent dynamo.{backend}
Expand Down Expand Up @@ -100,6 +117,46 @@ def _sidecar_worker_gpu_env(backend: str) -> dict[str, str]:
chat_payload_default(),
],
),
# Prefill/decode handoff is a critical native-sidecar path.
"vllm_disaggregated": EngineConfig(
name="vllm_disaggregated",
directory=vllm_sidecar_dir,
script_name="disagg.sh",
marks=[
pytest.mark.vllm,
pytest.mark.gpu_1,
pytest.mark.pre_merge,
pytest.mark.post_merge,
pytest.mark.nightly,
pytest.mark.timeout(1200),
pytest.mark.requested_vllm_kv_cache_bytes(1119388000),
],
model="Qwen/Qwen3-0.6B",
health_check_workers=True,
health_check_worker_count=2,
env={"PYTHONUNBUFFERED": "1", "MAX_MODEL_LEN": "2048"},
request_payloads=[_disaggregated_chat_payload()],
),
"sglang_disaggregated": EngineConfig(
name="sglang_disaggregated",
directory=sglang_sidecar_dir,
script_name="disagg.sh",
script_args=["--disable-cuda-graph"],
marks=[
pytest.mark.sglang,
pytest.mark.gpu_1,
pytest.mark.pre_merge,
pytest.mark.post_merge,
pytest.mark.nightly,
pytest.mark.timeout(1200),
pytest.mark.requested_sglang_kv_tokens(2048),
],
model="Qwen/Qwen3-0.6B",
health_check_workers=True,
health_check_worker_count=2,
env={"PYTHONUNBUFFERED": "1", "MAX_MODEL_LEN": "2048"},
request_payloads=[_disaggregated_chat_payload()],
),
}


Expand All @@ -120,19 +177,51 @@ def test_serve_deployment(
dynamo_dynamic_ports,
num_system_ports,
predownload_models,
monkeypatch,
):
"""
Launch a lib/sidecar/<backend>/launch/agg.sh script end-to-end (Dynamo
frontend + native-gRPC engine + dynamo-<backend>-sidecar) and confirm it
serves a real chat completion.
"""
"""Launch a native engine and sidecar deployment and validate chat completion."""
assert (
num_system_ports >= 2
), "serve tests require at least SYSTEM_PORT1 + SYSTEM_PORT2"
config = dataclasses.replace(
sidecar_config_test, frontend_port=dynamo_dynamic_ports.frontend_port
)
if config.name == "vllm_aggregated":
if config.name.endswith("_disaggregated"):
monkeypatch.delenv("DYN_NAMESPACE_WORKER_SUFFIX", raising=False)
monkeypatch.setenv("DYN_REQUEST_PLANE", "tcp")
backend = config.name.removesuffix("_disaggregated")
roles = ("DECODE", "PREFILL") if backend == "vllm" else ("PREFILL", "DECODE")
device = map_cuda_visible_devices([0], os.environ.get("CUDA_VISIBLE_DEVICES"))
assert device != "-1", "One visible GPU is required"
engine_env = {
**{f"{backend.upper()}_{role}_GPU": device for role in roles},
"DYN_NAMESPACE": f"sidecar-disagg-{generate_random_suffix()}",
"MODEL": config.model,
}
num_engine_ports = {"vllm": 4, "sglang": 5}[backend]
with reserved_ports(
num_engine_ports, start_port=DynamoPortRange.SERVE.value
) as engine_ports:
for index, role in enumerate(roles):
prefix = f"{backend.upper()}_{role}"
engine_env[f"{prefix}_HTTP_PORT"] = str(engine_ports[index * 2])
engine_env[f"{prefix}_GRPC_PORT"] = str(engine_ports[index * 2 + 1])
if backend == "vllm":
engine_env[f"{prefix}_NIXL_SIDE_CHANNEL_PORT"] = str(
dynamo_dynamic_ports.nixl_side_channel_ports[index]
)
if backend == "vllm":
engine_env["VLLM_PREFILL_KV_EVENT_PORT"] = str(
dynamo_dynamic_ports.kv_event_ports[1]
)
elif backend == "sglang":
engine_env["SGLANG_DISAGGREGATION_BOOTSTRAP_PORT"] = str(
engine_ports[4]
)
run_serve_deployment(
config, request, ports=dynamo_dynamic_ports, extra_env=engine_env
)
elif config.name == "vllm_aggregated":
with reserved_ports(2, start_port=DynamoPortRange.SERVE.value) as engine_ports:
run_serve_deployment(
config,
Expand Down
42 changes: 41 additions & 1 deletion tests/utils/payloads.py
Original file line number Diff line number Diff line change
Expand Up @@ -33,7 +33,11 @@
from tests.utils.http_checks import check_health_generate as check_health_generate
from tests.utils.http_checks import check_models_api as check_models_api
from tests.utils.prometheus import find_metric_samples, sum_metric_samples
from tests.utils.router_nvext import RouterNvextExpectation, validate_router_nvext
from tests.utils.router_nvext import (
RouterNvextExpectation,
require_router_worker_id,
validate_router_nvext,
)

logger = logging.getLogger(__name__)

Expand Down Expand Up @@ -215,6 +219,42 @@ def validate(self, response: Any, content: str) -> None:
)


class DisaggregatedChatPayload(ChatPayload):
"""Require a completed chat request served by distinct prefill and decode workers."""

def validate(self, response: Any, content: str) -> None:
super().validate(response, content)
result = response.json()
choices = result["choices"]
if len(choices) != 1:
raise AssertionError(f"Expected one completion, got {choices!r}")
if not isinstance(content, str) or not content.strip():
raise AssertionError("Completion is empty")
if choices[0].get("finish_reason") not in {"stop", "length"}:
raise AssertionError(f"Unexpected finish reason: {choices[0]!r}")

usage = result.get("usage")
if not isinstance(usage, dict):
raise AssertionError(f"Missing usage: {result!r}")
prompt_tokens = usage.get("prompt_tokens")
completion_tokens = usage.get("completion_tokens")
if type(prompt_tokens) is not int or prompt_tokens <= 0:
raise AssertionError(f"Expected positive prompt usage: {usage!r}")
if type(completion_tokens) is not int or completion_tokens <= 1:
raise AssertionError(
f"Expected decode to generate more than the prefill token: {usage!r}"
)

workers = require_router_worker_id(result, context=type(self).__name__)
for role in ("prefill_worker_id", "decode_worker_id"):
if type(workers.get(role)) is not int or workers[role] < 0:
raise AssertionError(f"Expected a valid {role}: {dict(workers)!r}")
if workers["prefill_worker_id"] == workers["decode_worker_id"]:
raise AssertionError(
f"Expected distinct prefill and decode workers: {dict(workers)!r}"
)


class RouterNvextChatPayload(ChatPayload):
"""Chat payload that validates structured router metadata in nvext."""

Expand Down
Loading