Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 2 additions & 4 deletions .github/workflows/nightly-ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -995,7 +995,7 @@ jobs:
dd_flaky_retry_enabled: ${{ 'false' }}
test_suite_name: trtllm
framework: trtllm
test_type: Sidecar E2E Test
test_type: TensorRT-LLM Sidecar 1-GPU E2E Test
amd_runner: prod-tester-amd-gpu-v2
target_tag_plain: ${{ needs.trtllm-build.outputs.target_tag_plain }}
cuda_version: '["13.1"]'
Expand Down Expand Up @@ -1027,9 +1027,7 @@ jobs:
enable_coverage: true
run_cpu_only_tests: true
cpu_only_test_markers: vllm and gpu_0
# not sidecar: sidecar-marked tests need the dynamo-vllm-sidecar binary
# installed (see pr.yaml's sidecar-vllm-test), which this plain
# vllm-runtime image doesn't have.
# Plain runtime images do not include sidecar binaries.
# TODO: this ad-hoc exclusion is duplicated across pr.yaml, post-merge-ci.yml,
# and nightly-ci.yml's gpu_test_markers; consolidate into shared-test.yml
# once there's a second marker that needs the same treatment.
Expand Down
29 changes: 7 additions & 22 deletions .github/workflows/post-merge-ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -217,9 +217,6 @@ jobs:
extra_tags: |
${{ github.ref_name == 'main' && 'main-vllm-runtime' || '' }}
${{ github.ref_name == 'main' && format('main-vllm-runtime-{0}', github.sha) || '' }}
# Lets sidecar-vllm-test (pr.yaml) fall back to this floating tag via its
# source_ref input instead of rebuilding vllm-runtime on every
# lib/sidecar/**-only PR.
test_extra_tags: |
${{ github.ref_name == 'main' && 'main-vllm-runtime-test' || '' }}
archive_sources: false
Expand Down Expand Up @@ -287,9 +284,6 @@ jobs:
extra_tags: |
${{ github.ref_name == 'main' && 'main-sglang-runtime' || '' }}
${{ github.ref_name == 'main' && format('main-sglang-runtime-{0}', github.sha) || '' }}
# Lets sidecar-sglang-test (pr.yaml) fall back to this floating tag via
# its source_ref input instead of rebuilding sglang-runtime on every
# lib/sidecar/**-only PR.
test_extra_tags: |
${{ github.ref_name == 'main' && 'main-sglang-runtime-test' || '' }}
archive_sources: false
Expand Down Expand Up @@ -357,9 +351,6 @@ jobs:
extra_tags: |
${{ github.ref_name == 'main' && 'main-trtllm-runtime' || '' }}
${{ github.ref_name == 'main' && format('main-trtllm-runtime-{0}', github.sha) || '' }}
# Lets sidecar-trtllm-test (pr.yaml) fall back to this floating tag via
# its source_ref input instead of rebuilding trtllm-runtime on every
# lib/sidecar/**-only PR.
test_extra_tags: |
${{ github.ref_name == 'main' && 'main-trtllm-runtime-test' || '' }}
archive_sources: false
Expand Down Expand Up @@ -491,7 +482,7 @@ jobs:
cuda_version: '["13.0"]'
platform: '["amd64"]'
run_sanity_check: false
gpu_test_markers: ${{ format('(pre_merge or post_merge) and sidecar and vllm and gpu_{0}', matrix.gpu_count) }}
gpu_test_markers: ${{ format('post_merge and sidecar and vllm and gpu_{0}', matrix.gpu_count) }}
gpu_test_timeout_minutes: 45
sidecar_backend: vllm
sidecar_binary_artifact: ${{ needs.sidecar-build.outputs.binary_artifact }}
Expand All @@ -515,7 +506,7 @@ jobs:
cuda_version: '["13.0"]'
platform: '["amd64"]'
run_sanity_check: false
gpu_test_markers: ${{ format('(pre_merge or post_merge) and sidecar and sglang and gpu_{0}', matrix.gpu_count) }}
gpu_test_markers: ${{ format('post_merge and sidecar and sglang and gpu_{0}', matrix.gpu_count) }}
gpu_test_timeout_minutes: 45
sidecar_backend: sglang
sidecar_binary_artifact: ${{ needs.sidecar-build.outputs.binary_artifact }}
Expand All @@ -529,13 +520,13 @@ jobs:
dd_env: post-merge
test_suite_name: trtllm
framework: trtllm
test_type: Sidecar E2E Test
test_type: TensorRT-LLM Sidecar 1-GPU E2E Test
amd_runner: prod-tester-amd-gpu-v2
target_tag_plain: ${{ needs.trtllm-build.outputs.target_tag_plain }}
cuda_version: '["13.1"]'
platform: '["amd64"]'
run_sanity_check: false
gpu_test_markers: '(pre_merge or post_merge) and sidecar and trtllm and gpu_1'
gpu_test_markers: 'post_merge and sidecar and trtllm and gpu_1'
gpu_test_timeout_minutes: 45
sidecar_backend: trtllm
sidecar_binary_artifact: ${{ needs.sidecar-build.outputs.binary_artifact }}
Expand All @@ -556,9 +547,7 @@ jobs:
platform: '["amd64", "arm64"]' # arm64 for CPU tests, single GPU tests are skipped
run_cpu_only_tests: true
cpu_only_test_markers: '(pre_merge or post_merge) and vllm and gpu_0'
# not sidecar: sidecar-marked tests need dynamo-vllm-sidecar layered onto
# the image (see pr.yaml's sidecar-vllm-test), which this plain
# vllm-runtime image doesn't have.
# Plain runtime images do not include sidecar binaries.
gpu_test_markers: '(pre_merge or post_merge) and vllm and gpu_1 and not sidecar'
gpu_test_timeout_minutes: 120
# Profiled tests (those carrying @pytest.mark.profiled_vram_gib(N)) run concurrently in
Expand Down Expand Up @@ -637,9 +626,7 @@ jobs:
platform: '["amd64", "arm64"]' # arm64 for CPU tests, single GPU tests are skipped
run_cpu_only_tests: true
cpu_only_test_markers: '(pre_merge or post_merge) and sglang and gpu_0'
# not sidecar: sidecar-marked tests need dynamo-sglang-sidecar layered
# onto the image (see pr.yaml's sidecar-sglang-test), which this plain
# sglang-runtime image doesn't have.
# Plain runtime images do not include sidecar binaries.
gpu_test_markers: '(pre_merge or post_merge) and sglang and gpu_1 and not sidecar'
gpu_test_timeout_minutes: 120
# Profiled tests run in the VRAM-aware GPU stage; unprofiled fall through to sequential.
Expand Down Expand Up @@ -682,9 +669,7 @@ jobs:
platform: '["amd64", "arm64"]' # arm64 for CPU tests, single GPU tests are skipped
run_cpu_only_tests: true
cpu_only_test_markers: '(pre_merge or post_merge) and trtllm and gpu_0'
# not sidecar: sidecar-marked tests need dynamo-trtllm-sidecar layered
# onto the image (see pr.yaml's sidecar-trtllm-test), which this plain
# trtllm-runtime image doesn't have.
# Plain runtime images do not include sidecar binaries.
gpu_test_markers: '(pre_merge or post_merge) and trtllm and gpu_1 and not sidecar'
gpu_test_timeout_minutes: 120
# Profiled tests run in the VRAM-aware GPU stage; unprofiled fall through to sequential.
Expand Down
105 changes: 3 additions & 102 deletions .github/workflows/pr.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -131,9 +131,6 @@ jobs:
- helm-chart-tests
- power-agent
- sidecar-build
- sidecar-vllm-test
- sidecar-sglang-test
- sidecar-trtllm-test
- vllm-build
- vllm-local-dev-build
- sglang-local-dev-build
Expand Down Expand Up @@ -363,96 +360,6 @@ jobs:
diff_base_branch: ${{ needs.changed-files.outputs.base_ref }}
diff_merge_base_sha: ${{ needs.changed-files.outputs.merge_base_sha }}

# Reuse the backend test image and install the sidecar artifact before pytest.

sidecar-vllm-test:
name: sidecar-runtime # This name overlaps with other sidecar jobs to group them in the UI
needs: [changed-files, vllm-build, sidecar-build]
strategy:
fail-fast: false
matrix:
gpu_count: ${{ fromJSON(needs.changed-files.outputs.run_multigpu_tests == 'true' && '[1,2]' || '[1]') }}
# vllm-build must have succeeded or been intentionally skipped (a
# lib/sidecar-only PR) -- not failed or cancelled -- or the fallback below
# would silently test this PR's sidecar binary against main's vllm image
# and report green while this PR's own vllm-runtime build is red.
if: ${{ !cancelled() && needs.changed-files.outputs.sidecar == 'true' && needs.sidecar-build.result == 'success' && (needs.vllm-build.result == 'success' || needs.vllm-build.result == 'skipped') }}
permissions:
actions: read
contents: read
uses: $/.github/workflows/shared-test.yml
with:
test_suite_name: vllm
framework: vllm # matches the framework passed to vllm-build
test_type: ${{ format('Sidecar {0}-GPU E2E Test', matrix.gpu_count) }}
amd_runner: ${{ matrix.gpu_count == 2 && 'prod-tester-amd-gpu-2-v2' || 'prod-tester-amd-gpu-v2' }}
target_tag_plain: ${{ needs.vllm-build.result == 'success' && needs.vllm-build.outputs.target_tag_plain || 'vllm-runtime' }}
source_ref: ${{ needs.vllm-build.result == 'success' && github.sha || 'main' }}
cuda_version: '["13.0"]'
platform: '["amd64"]'
run_sanity_check: false
gpu_test_markers: ${{ format('pre_merge and sidecar and vllm and gpu_{0}', matrix.gpu_count) }}
gpu_test_timeout_minutes: 45
sidecar_backend: vllm
sidecar_binary_artifact: ${{ needs.sidecar-build.outputs.binary_artifact }}
secrets: inherit

sidecar-sglang-test:
name: sidecar-runtime # This name overlaps with other sidecar jobs to group them in the UI
needs: [changed-files, sglang-build, sidecar-build]
strategy:
fail-fast: false
matrix:
gpu_count: ${{ fromJSON(needs.changed-files.outputs.run_multigpu_tests == 'true' && '[1,2]' || '[1]') }}
# See sidecar-vllm-test above: sglang-build must have succeeded or been
# intentionally skipped, not failed or cancelled.
if: ${{ !cancelled() && needs.changed-files.outputs.sidecar == 'true' && needs.sidecar-build.result == 'success' && (needs.sglang-build.result == 'success' || needs.sglang-build.result == 'skipped') }}
permissions:
actions: read
contents: read
uses: $/.github/workflows/shared-test.yml
with:
test_suite_name: sglang
framework: sglang # matches the framework passed to sglang-build
test_type: ${{ format('Sidecar {0}-GPU E2E Test', matrix.gpu_count) }}
amd_runner: ${{ matrix.gpu_count == 2 && 'prod-tester-amd-gpu-2-v2' || 'prod-tester-amd-gpu-v2' }}
target_tag_plain: ${{ needs.sglang-build.result == 'success' && needs.sglang-build.outputs.target_tag_plain || 'sglang-runtime' }}
source_ref: ${{ needs.sglang-build.result == 'success' && github.sha || 'main' }}
cuda_version: '["13.0"]'
platform: '["amd64"]'
run_sanity_check: false
gpu_test_markers: ${{ format('pre_merge and sidecar and sglang and gpu_{0}', matrix.gpu_count) }}
gpu_test_timeout_minutes: 45
sidecar_backend: sglang
sidecar_binary_artifact: ${{ needs.sidecar-build.outputs.binary_artifact }}
secrets: inherit

sidecar-trtllm-test:
name: sidecar-runtime # This name overlaps with other sidecar jobs to group them in the UI
needs: [changed-files, trtllm-build, sidecar-build]
# See sidecar-vllm-test above: trtllm-build must have succeeded or been
# intentionally skipped, not failed or cancelled.
if: ${{ !cancelled() && needs.changed-files.outputs.sidecar == 'true' && needs.sidecar-build.result == 'success' && (needs.trtllm-build.result == 'success' || needs.trtllm-build.result == 'skipped') }}
permissions:
actions: read
contents: read
uses: $/.github/workflows/shared-test.yml
with:
test_suite_name: trtllm
framework: trtllm # matches the framework passed to trtllm-build
test_type: Sidecar E2E Test
amd_runner: prod-tester-amd-gpu-v2
target_tag_plain: ${{ needs.trtllm-build.result == 'success' && needs.trtllm-build.outputs.target_tag_plain || 'trtllm-runtime' }}
source_ref: ${{ needs.trtllm-build.result == 'success' && github.sha || 'main' }}
cuda_version: '["13.1"]'
platform: '["amd64"]'
run_sanity_check: false
gpu_test_markers: pre_merge and sidecar and trtllm and gpu_1
gpu_test_timeout_minutes: 45
sidecar_backend: trtllm
sidecar_binary_artifact: ${{ needs.sidecar-build.outputs.binary_artifact }}
secrets: inherit

# ============================================================================
# FRAMEWORK PIPELINES (Build → Test → Copy)
# ============================================================================
Expand Down Expand Up @@ -908,9 +815,7 @@ jobs:
# Keep the outer step budget above the test's 450s timeout so pytest can
# finish teardown and upload diagnostics instead of being killed first.
cpu_e2e_test_timeout_minutes: 15
# not sidecar: sidecar-marked tests need dynamo-vllm-sidecar layered onto
# the image (see sidecar-vllm-test below), which this plain vllm-runtime
# image doesn't have.
# Plain runtime images do not include sidecar binaries.
gpu_test_markers: pre_merge and vllm and gpu_1 and not sidecar
gpu_test_timeout_minutes: 45
# Profiled tests run in the parallel stage; unprofiled fall through to sequential.
Expand Down Expand Up @@ -956,9 +861,7 @@ jobs:
platform: '["amd64", "arm64"]' # arm64 for CPU tests, single GPU tests are skipped
run_cpu_only_tests: true
cpu_only_test_markers: pre_merge and sglang and gpu_0
# not sidecar: sidecar-marked tests need dynamo-sglang-sidecar layered
# onto the image (see sidecar-sglang-test below), which this plain
# sglang-runtime image doesn't have.
# Plain runtime images do not include sidecar binaries.
gpu_test_markers: pre_merge and sglang and gpu_1 and not sidecar
# Profiled tests run in the VRAM-aware GPU stage; unprofiled fall through to sequential.
# Current single-GPU runners are 24 GiB, so this cap admits the profiled SGLang pool
Expand Down Expand Up @@ -1004,9 +907,7 @@ jobs:
platform: '["amd64", "arm64"]' # arm64 for CPU tests, single GPU tests are skipped
run_cpu_only_tests: true
cpu_only_test_markers: pre_merge and trtllm and gpu_0
# not sidecar: sidecar-marked tests need dynamo-trtllm-sidecar layered
# onto the image (see sidecar-trtllm-test below), which this plain
# trtllm-runtime image doesn't have.
# Plain runtime images do not include sidecar binaries.
gpu_test_markers: pre_merge and trtllm and gpu_1 and not sidecar
# Profiled tests run in the VRAM-aware GPU stage; unprofiled fall through to sequential.
# Current single-GPU runners are 24 GiB, so this cap admits the profiled TRT-LLM pool
Expand Down
15 changes: 5 additions & 10 deletions tests/serve/test_sidecar.py
Original file line number Diff line number Diff line change
Expand Up @@ -110,7 +110,7 @@ def _disaggregated_chat_payload() -> DisaggregatedChatPayload:
pytest.mark.gpu_1,
# Let the 600s health check report failure before pytest times out.
pytest.mark.timeout(780),
pytest.mark.pre_merge,
pytest.mark.post_merge,
],
model="Qwen/Qwen3-0.6B",
# Flush Python output promptly into CI logs.
Expand All @@ -127,7 +127,7 @@ def _disaggregated_chat_payload() -> DisaggregatedChatPayload:
pytest.mark.sglang,
pytest.mark.gpu_1,
pytest.mark.timeout(780),
pytest.mark.pre_merge,
pytest.mark.post_merge,
],
model="Qwen/Qwen3-0.6B",
env={"PYTHONUNBUFFERED": "1"},
Expand All @@ -143,7 +143,7 @@ def _disaggregated_chat_payload() -> DisaggregatedChatPayload:
pytest.mark.trtllm,
pytest.mark.gpu_1,
pytest.mark.timeout(780),
pytest.mark.pre_merge,
pytest.mark.post_merge,
pytest.mark.skipif(
not _trtllm_serves_openengine(),
reason=TRTLLM_OPENENGINE_SKIP_REASON,
Expand All @@ -160,7 +160,6 @@ def _disaggregated_chat_payload() -> DisaggregatedChatPayload:
chat_payload_default(),
],
),
# Prefill/decode handoff is a critical native-sidecar path.
"trtllm_disaggregated": EngineConfig(
name="trtllm_disaggregated",
directory=trtllm_sidecar_dir,
Expand All @@ -184,7 +183,7 @@ def _disaggregated_chat_payload() -> DisaggregatedChatPayload:
# Leaving that at its 600s default would let the health check fail
# at the single-engine budget and then idle until the kill timer.
pytest.mark.timeout(1200),
pytest.mark.pre_merge,
pytest.mark.post_merge,
pytest.mark.skipif(
not _trtllm_serves_openengine(),
reason=TRTLLM_OPENENGINE_SKIP_REASON,
Expand Down Expand Up @@ -213,9 +212,7 @@ def _disaggregated_chat_payload() -> DisaggregatedChatPayload:
marks=[
pytest.mark.vllm,
pytest.mark.gpu_1,
pytest.mark.pre_merge,
pytest.mark.post_merge,
pytest.mark.nightly,
pytest.mark.timeout(1200),
pytest.mark.requested_vllm_kv_cache_bytes(1119388000),
],
Expand All @@ -233,9 +230,7 @@ def _disaggregated_chat_payload() -> DisaggregatedChatPayload:
marks=[
pytest.mark.sglang,
pytest.mark.gpu_1,
pytest.mark.pre_merge,
pytest.mark.post_merge,
pytest.mark.nightly,
pytest.mark.timeout(1200),
pytest.mark.requested_sglang_kv_tokens(2048),
],
Expand Down Expand Up @@ -331,7 +326,7 @@ def test_serve_deployment(
@pytest.mark.router
@pytest.mark.sidecar
@pytest.mark.e2e
@pytest.mark.pre_merge # Guard native KV-event discovery on every sidecar change.
@pytest.mark.post_merge
Comment thread
JulienDarve marked this conversation as resolved.
@pytest.mark.timeout(1200)
@pytest.mark.parametrize("request_plane", ["tcp"], indirect=True)
@pytest.mark.parametrize(
Expand Down
Loading