diff --git a/.github/workflows/nightly-ci.yml b/.github/workflows/nightly-ci.yml index 70165b26ec5d..7e376453d382 100644 --- a/.github/workflows/nightly-ci.yml +++ b/.github/workflows/nightly-ci.yml @@ -995,7 +995,7 @@ jobs: dd_flaky_retry_enabled: ${{ 'false' }} test_suite_name: trtllm framework: trtllm - test_type: Sidecar E2E Test + test_type: TensorRT-LLM Sidecar 1-GPU E2E Test amd_runner: prod-tester-amd-gpu-v2 target_tag_plain: ${{ needs.trtllm-build.outputs.target_tag_plain }} cuda_version: '["13.1"]' @@ -1027,9 +1027,7 @@ jobs: enable_coverage: true run_cpu_only_tests: true cpu_only_test_markers: vllm and gpu_0 - # not sidecar: sidecar-marked tests need the dynamo-vllm-sidecar binary - # installed (see pr.yaml's sidecar-vllm-test), which this plain - # vllm-runtime image doesn't have. + # Plain runtime images do not include sidecar binaries. # TODO: this ad-hoc exclusion is duplicated across pr.yaml, post-merge-ci.yml, # and nightly-ci.yml's gpu_test_markers; consolidate into shared-test.yml # once there's a second marker that needs the same treatment. diff --git a/.github/workflows/post-merge-ci.yml b/.github/workflows/post-merge-ci.yml index 4f13f90db936..6be5e1d30dbb 100644 --- a/.github/workflows/post-merge-ci.yml +++ b/.github/workflows/post-merge-ci.yml @@ -217,9 +217,6 @@ jobs: extra_tags: | ${{ github.ref_name == 'main' && 'main-vllm-runtime' || '' }} ${{ github.ref_name == 'main' && format('main-vllm-runtime-{0}', github.sha) || '' }} - # Lets sidecar-vllm-test (pr.yaml) fall back to this floating tag via its - # source_ref input instead of rebuilding vllm-runtime on every - # lib/sidecar/**-only PR. test_extra_tags: | ${{ github.ref_name == 'main' && 'main-vllm-runtime-test' || '' }} archive_sources: false @@ -287,9 +284,6 @@ jobs: extra_tags: | ${{ github.ref_name == 'main' && 'main-sglang-runtime' || '' }} ${{ github.ref_name == 'main' && format('main-sglang-runtime-{0}', github.sha) || '' }} - # Lets sidecar-sglang-test (pr.yaml) fall back to this floating tag via - # its source_ref input instead of rebuilding sglang-runtime on every - # lib/sidecar/**-only PR. test_extra_tags: | ${{ github.ref_name == 'main' && 'main-sglang-runtime-test' || '' }} archive_sources: false @@ -357,9 +351,6 @@ jobs: extra_tags: | ${{ github.ref_name == 'main' && 'main-trtllm-runtime' || '' }} ${{ github.ref_name == 'main' && format('main-trtllm-runtime-{0}', github.sha) || '' }} - # Lets sidecar-trtllm-test (pr.yaml) fall back to this floating tag via - # its source_ref input instead of rebuilding trtllm-runtime on every - # lib/sidecar/**-only PR. test_extra_tags: | ${{ github.ref_name == 'main' && 'main-trtllm-runtime-test' || '' }} archive_sources: false @@ -491,7 +482,7 @@ jobs: cuda_version: '["13.0"]' platform: '["amd64"]' run_sanity_check: false - gpu_test_markers: ${{ format('(pre_merge or post_merge) and sidecar and vllm and gpu_{0}', matrix.gpu_count) }} + gpu_test_markers: ${{ format('post_merge and sidecar and vllm and gpu_{0}', matrix.gpu_count) }} gpu_test_timeout_minutes: 45 sidecar_backend: vllm sidecar_binary_artifact: ${{ needs.sidecar-build.outputs.binary_artifact }} @@ -515,7 +506,7 @@ jobs: cuda_version: '["13.0"]' platform: '["amd64"]' run_sanity_check: false - gpu_test_markers: ${{ format('(pre_merge or post_merge) and sidecar and sglang and gpu_{0}', matrix.gpu_count) }} + gpu_test_markers: ${{ format('post_merge and sidecar and sglang and gpu_{0}', matrix.gpu_count) }} gpu_test_timeout_minutes: 45 sidecar_backend: sglang sidecar_binary_artifact: ${{ needs.sidecar-build.outputs.binary_artifact }} @@ -529,13 +520,13 @@ jobs: dd_env: post-merge test_suite_name: trtllm framework: trtllm - test_type: Sidecar E2E Test + test_type: TensorRT-LLM Sidecar 1-GPU E2E Test amd_runner: prod-tester-amd-gpu-v2 target_tag_plain: ${{ needs.trtllm-build.outputs.target_tag_plain }} cuda_version: '["13.1"]' platform: '["amd64"]' run_sanity_check: false - gpu_test_markers: '(pre_merge or post_merge) and sidecar and trtllm and gpu_1' + gpu_test_markers: 'post_merge and sidecar and trtllm and gpu_1' gpu_test_timeout_minutes: 45 sidecar_backend: trtllm sidecar_binary_artifact: ${{ needs.sidecar-build.outputs.binary_artifact }} @@ -556,9 +547,7 @@ jobs: platform: '["amd64", "arm64"]' # arm64 for CPU tests, single GPU tests are skipped run_cpu_only_tests: true cpu_only_test_markers: '(pre_merge or post_merge) and vllm and gpu_0' - # not sidecar: sidecar-marked tests need dynamo-vllm-sidecar layered onto - # the image (see pr.yaml's sidecar-vllm-test), which this plain - # vllm-runtime image doesn't have. + # Plain runtime images do not include sidecar binaries. gpu_test_markers: '(pre_merge or post_merge) and vllm and gpu_1 and not sidecar' gpu_test_timeout_minutes: 120 # Profiled tests (those carrying @pytest.mark.profiled_vram_gib(N)) run concurrently in @@ -637,9 +626,7 @@ jobs: platform: '["amd64", "arm64"]' # arm64 for CPU tests, single GPU tests are skipped run_cpu_only_tests: true cpu_only_test_markers: '(pre_merge or post_merge) and sglang and gpu_0' - # not sidecar: sidecar-marked tests need dynamo-sglang-sidecar layered - # onto the image (see pr.yaml's sidecar-sglang-test), which this plain - # sglang-runtime image doesn't have. + # Plain runtime images do not include sidecar binaries. gpu_test_markers: '(pre_merge or post_merge) and sglang and gpu_1 and not sidecar' gpu_test_timeout_minutes: 120 # Profiled tests run in the VRAM-aware GPU stage; unprofiled fall through to sequential. @@ -682,9 +669,7 @@ jobs: platform: '["amd64", "arm64"]' # arm64 for CPU tests, single GPU tests are skipped run_cpu_only_tests: true cpu_only_test_markers: '(pre_merge or post_merge) and trtllm and gpu_0' - # not sidecar: sidecar-marked tests need dynamo-trtllm-sidecar layered - # onto the image (see pr.yaml's sidecar-trtllm-test), which this plain - # trtllm-runtime image doesn't have. + # Plain runtime images do not include sidecar binaries. gpu_test_markers: '(pre_merge or post_merge) and trtllm and gpu_1 and not sidecar' gpu_test_timeout_minutes: 120 # Profiled tests run in the VRAM-aware GPU stage; unprofiled fall through to sequential. diff --git a/.github/workflows/pr.yaml b/.github/workflows/pr.yaml index c994e522e2aa..376b6690d76a 100644 --- a/.github/workflows/pr.yaml +++ b/.github/workflows/pr.yaml @@ -131,9 +131,6 @@ jobs: - helm-chart-tests - power-agent - sidecar-build - - sidecar-vllm-test - - sidecar-sglang-test - - sidecar-trtllm-test - vllm-build - vllm-local-dev-build - sglang-local-dev-build @@ -363,96 +360,6 @@ jobs: diff_base_branch: ${{ needs.changed-files.outputs.base_ref }} diff_merge_base_sha: ${{ needs.changed-files.outputs.merge_base_sha }} - # Reuse the backend test image and install the sidecar artifact before pytest. - - sidecar-vllm-test: - name: sidecar-runtime # This name overlaps with other sidecar jobs to group them in the UI - needs: [changed-files, vllm-build, sidecar-build] - strategy: - fail-fast: false - matrix: - gpu_count: ${{ fromJSON(needs.changed-files.outputs.run_multigpu_tests == 'true' && '[1,2]' || '[1]') }} - # vllm-build must have succeeded or been intentionally skipped (a - # lib/sidecar-only PR) -- not failed or cancelled -- or the fallback below - # would silently test this PR's sidecar binary against main's vllm image - # and report green while this PR's own vllm-runtime build is red. - if: ${{ !cancelled() && needs.changed-files.outputs.sidecar == 'true' && needs.sidecar-build.result == 'success' && (needs.vllm-build.result == 'success' || needs.vllm-build.result == 'skipped') }} - permissions: - actions: read - contents: read - uses: $/.github/workflows/shared-test.yml - with: - test_suite_name: vllm - framework: vllm # matches the framework passed to vllm-build - test_type: ${{ format('Sidecar {0}-GPU E2E Test', matrix.gpu_count) }} - amd_runner: ${{ matrix.gpu_count == 2 && 'prod-tester-amd-gpu-2-v2' || 'prod-tester-amd-gpu-v2' }} - target_tag_plain: ${{ needs.vllm-build.result == 'success' && needs.vllm-build.outputs.target_tag_plain || 'vllm-runtime' }} - source_ref: ${{ needs.vllm-build.result == 'success' && github.sha || 'main' }} - cuda_version: '["13.0"]' - platform: '["amd64"]' - run_sanity_check: false - gpu_test_markers: ${{ format('pre_merge and sidecar and vllm and gpu_{0}', matrix.gpu_count) }} - gpu_test_timeout_minutes: 45 - sidecar_backend: vllm - sidecar_binary_artifact: ${{ needs.sidecar-build.outputs.binary_artifact }} - secrets: inherit - - sidecar-sglang-test: - name: sidecar-runtime # This name overlaps with other sidecar jobs to group them in the UI - needs: [changed-files, sglang-build, sidecar-build] - strategy: - fail-fast: false - matrix: - gpu_count: ${{ fromJSON(needs.changed-files.outputs.run_multigpu_tests == 'true' && '[1,2]' || '[1]') }} - # See sidecar-vllm-test above: sglang-build must have succeeded or been - # intentionally skipped, not failed or cancelled. - if: ${{ !cancelled() && needs.changed-files.outputs.sidecar == 'true' && needs.sidecar-build.result == 'success' && (needs.sglang-build.result == 'success' || needs.sglang-build.result == 'skipped') }} - permissions: - actions: read - contents: read - uses: $/.github/workflows/shared-test.yml - with: - test_suite_name: sglang - framework: sglang # matches the framework passed to sglang-build - test_type: ${{ format('Sidecar {0}-GPU E2E Test', matrix.gpu_count) }} - amd_runner: ${{ matrix.gpu_count == 2 && 'prod-tester-amd-gpu-2-v2' || 'prod-tester-amd-gpu-v2' }} - target_tag_plain: ${{ needs.sglang-build.result == 'success' && needs.sglang-build.outputs.target_tag_plain || 'sglang-runtime' }} - source_ref: ${{ needs.sglang-build.result == 'success' && github.sha || 'main' }} - cuda_version: '["13.0"]' - platform: '["amd64"]' - run_sanity_check: false - gpu_test_markers: ${{ format('pre_merge and sidecar and sglang and gpu_{0}', matrix.gpu_count) }} - gpu_test_timeout_minutes: 45 - sidecar_backend: sglang - sidecar_binary_artifact: ${{ needs.sidecar-build.outputs.binary_artifact }} - secrets: inherit - - sidecar-trtllm-test: - name: sidecar-runtime # This name overlaps with other sidecar jobs to group them in the UI - needs: [changed-files, trtllm-build, sidecar-build] - # See sidecar-vllm-test above: trtllm-build must have succeeded or been - # intentionally skipped, not failed or cancelled. - if: ${{ !cancelled() && needs.changed-files.outputs.sidecar == 'true' && needs.sidecar-build.result == 'success' && (needs.trtllm-build.result == 'success' || needs.trtllm-build.result == 'skipped') }} - permissions: - actions: read - contents: read - uses: $/.github/workflows/shared-test.yml - with: - test_suite_name: trtllm - framework: trtllm # matches the framework passed to trtllm-build - test_type: Sidecar E2E Test - amd_runner: prod-tester-amd-gpu-v2 - target_tag_plain: ${{ needs.trtllm-build.result == 'success' && needs.trtllm-build.outputs.target_tag_plain || 'trtllm-runtime' }} - source_ref: ${{ needs.trtllm-build.result == 'success' && github.sha || 'main' }} - cuda_version: '["13.1"]' - platform: '["amd64"]' - run_sanity_check: false - gpu_test_markers: pre_merge and sidecar and trtllm and gpu_1 - gpu_test_timeout_minutes: 45 - sidecar_backend: trtllm - sidecar_binary_artifact: ${{ needs.sidecar-build.outputs.binary_artifact }} - secrets: inherit - # ============================================================================ # FRAMEWORK PIPELINES (Build → Test → Copy) # ============================================================================ @@ -908,9 +815,7 @@ jobs: # Keep the outer step budget above the test's 450s timeout so pytest can # finish teardown and upload diagnostics instead of being killed first. cpu_e2e_test_timeout_minutes: 15 - # not sidecar: sidecar-marked tests need dynamo-vllm-sidecar layered onto - # the image (see sidecar-vllm-test below), which this plain vllm-runtime - # image doesn't have. + # Plain runtime images do not include sidecar binaries. gpu_test_markers: pre_merge and vllm and gpu_1 and not sidecar gpu_test_timeout_minutes: 45 # Profiled tests run in the parallel stage; unprofiled fall through to sequential. @@ -956,9 +861,7 @@ jobs: platform: '["amd64", "arm64"]' # arm64 for CPU tests, single GPU tests are skipped run_cpu_only_tests: true cpu_only_test_markers: pre_merge and sglang and gpu_0 - # not sidecar: sidecar-marked tests need dynamo-sglang-sidecar layered - # onto the image (see sidecar-sglang-test below), which this plain - # sglang-runtime image doesn't have. + # Plain runtime images do not include sidecar binaries. gpu_test_markers: pre_merge and sglang and gpu_1 and not sidecar # Profiled tests run in the VRAM-aware GPU stage; unprofiled fall through to sequential. # Current single-GPU runners are 24 GiB, so this cap admits the profiled SGLang pool @@ -1004,9 +907,7 @@ jobs: platform: '["amd64", "arm64"]' # arm64 for CPU tests, single GPU tests are skipped run_cpu_only_tests: true cpu_only_test_markers: pre_merge and trtllm and gpu_0 - # not sidecar: sidecar-marked tests need dynamo-trtllm-sidecar layered - # onto the image (see sidecar-trtllm-test below), which this plain - # trtllm-runtime image doesn't have. + # Plain runtime images do not include sidecar binaries. gpu_test_markers: pre_merge and trtllm and gpu_1 and not sidecar # Profiled tests run in the VRAM-aware GPU stage; unprofiled fall through to sequential. # Current single-GPU runners are 24 GiB, so this cap admits the profiled TRT-LLM pool diff --git a/tests/serve/test_sidecar.py b/tests/serve/test_sidecar.py index 60adc1e2d3c3..b721a06538a2 100644 --- a/tests/serve/test_sidecar.py +++ b/tests/serve/test_sidecar.py @@ -110,7 +110,7 @@ def _disaggregated_chat_payload() -> DisaggregatedChatPayload: pytest.mark.gpu_1, # Let the 600s health check report failure before pytest times out. pytest.mark.timeout(780), - pytest.mark.pre_merge, + pytest.mark.post_merge, ], model="Qwen/Qwen3-0.6B", # Flush Python output promptly into CI logs. @@ -127,7 +127,7 @@ def _disaggregated_chat_payload() -> DisaggregatedChatPayload: pytest.mark.sglang, pytest.mark.gpu_1, pytest.mark.timeout(780), - pytest.mark.pre_merge, + pytest.mark.post_merge, ], model="Qwen/Qwen3-0.6B", env={"PYTHONUNBUFFERED": "1"}, @@ -143,7 +143,7 @@ def _disaggregated_chat_payload() -> DisaggregatedChatPayload: pytest.mark.trtllm, pytest.mark.gpu_1, pytest.mark.timeout(780), - pytest.mark.pre_merge, + pytest.mark.post_merge, pytest.mark.skipif( not _trtllm_serves_openengine(), reason=TRTLLM_OPENENGINE_SKIP_REASON, @@ -160,7 +160,6 @@ def _disaggregated_chat_payload() -> DisaggregatedChatPayload: chat_payload_default(), ], ), - # Prefill/decode handoff is a critical native-sidecar path. "trtllm_disaggregated": EngineConfig( name="trtllm_disaggregated", directory=trtllm_sidecar_dir, @@ -184,7 +183,7 @@ def _disaggregated_chat_payload() -> DisaggregatedChatPayload: # Leaving that at its 600s default would let the health check fail # at the single-engine budget and then idle until the kill timer. pytest.mark.timeout(1200), - pytest.mark.pre_merge, + pytest.mark.post_merge, pytest.mark.skipif( not _trtllm_serves_openengine(), reason=TRTLLM_OPENENGINE_SKIP_REASON, @@ -213,9 +212,7 @@ def _disaggregated_chat_payload() -> DisaggregatedChatPayload: marks=[ pytest.mark.vllm, pytest.mark.gpu_1, - pytest.mark.pre_merge, pytest.mark.post_merge, - pytest.mark.nightly, pytest.mark.timeout(1200), pytest.mark.requested_vllm_kv_cache_bytes(1119388000), ], @@ -233,9 +230,7 @@ def _disaggregated_chat_payload() -> DisaggregatedChatPayload: marks=[ pytest.mark.sglang, pytest.mark.gpu_1, - pytest.mark.pre_merge, pytest.mark.post_merge, - pytest.mark.nightly, pytest.mark.timeout(1200), pytest.mark.requested_sglang_kv_tokens(2048), ], @@ -331,7 +326,7 @@ def test_serve_deployment( @pytest.mark.router @pytest.mark.sidecar @pytest.mark.e2e -@pytest.mark.pre_merge # Guard native KV-event discovery on every sidecar change. +@pytest.mark.post_merge @pytest.mark.timeout(1200) @pytest.mark.parametrize("request_plane", ["tcp"], indirect=True) @pytest.mark.parametrize(