From 5dacf5918c557db7d5c83110cf04e543a25f282a Mon Sep 17 00:00:00 2001 From: key4ng Date: Fri, 27 Feb 2026 00:08:09 +0000 Subject: [PATCH 1/5] chore(ci): update GPU runner configuration and add new runner for H100 test environment - Changed runner name to '4-gpu-h100-test' in the CI workflow. - Added a new RunnerDeployment for 'arc-runner-gpu-h100-test' with specific resource configurations in the Kubernetes setup. - Commented out several test configurations for clarity and future reference. Signed-off-by: key4ng --- .github/workflows/pr-test-rust.yml | 124 +++++++++--------- .../k8s-runner-resources/arc-runner-gpu.yaml | 91 +++++++++++++ 2 files changed, 153 insertions(+), 62 deletions(-) diff --git a/.github/workflows/pr-test-rust.yml b/.github/workflows/pr-test-rust.yml index a2a151e0d4..3900cb60d2 100644 --- a/.github/workflows/pr-test-rust.yml +++ b/.github/workflows/pr-test-rust.yml @@ -253,68 +253,68 @@ jobs: upload_benchmarks: true parallel_opts: "" # No parallel for benchmarks (performance measurement) ignore_opts: "--ignore=e2e_test/benchmarks/test_go_bindings_perf.py --ignore=e2e_test/benchmarks/test_nightly_perf.py --ignore=e2e_test/benchmarks/test_pd_perf.py" # Go and nightly benchmarks run in dedicated jobs - runner: 4-gpu-h100 - - name: agentic-apis - timeout: 30 - test_dirs: "e2e_test/responses e2e_test/messages" - extra_deps: "" - env_vars: "SHOW_WORKER_LOGS=0 SHOW_ROUTER_LOGS=1" - reruns: "--reruns 2 --reruns-delay 5" - setup_agentic_deps: true - parallel_opts: "" - ignore_opts: "" - test_filter: "-m 'not external'" - - name: e2e - timeout: 20 - test_dirs: "e2e_test/router e2e_test/embeddings" - # py is needed for pytest-parallel; sentence-transformers provides HuggingFace reference embeddings - extra_deps: "pytest-parallel py sentence-transformers" - env_vars: "SHOW_WORKER_LOGS=0 SHOW_ROUTER_LOGS=1" - reruns: "--reruns 2 --reruns-delay 5" - parallel_opts: "--workers 1 --tests-per-worker 4" # Thread-based parallelism - ignore_opts: "" - - name: chat-completions-sglang - timeout: 20 - test_dirs: "e2e_test/chat_completions" - extra_deps: "pytest-parallel py" - env_vars: "E2E_RUNTIME=sglang SHOW_WORKER_LOGS=0 SHOW_ROUTER_LOGS=1" - reruns: "--reruns 2 --reruns-delay 5" - parallel_opts: "--workers 1 --tests-per-worker 4" - test_filter: "" - ignore_opts: "" - - name: chat-completions-vllm - timeout: 20 - test_dirs: "e2e_test/chat_completions" - extra_deps: "pytest-parallel py" - env_vars: "E2E_RUNTIME=vllm SHOW_WORKER_LOGS=0 SHOW_ROUTER_LOGS=1" - reruns: "--reruns 2 --reruns-delay 5" - parallel_opts: "--workers 1 --tests-per-worker 4" - # TODO: Remove filter when vLLM supports logprobs and n>1 with greedy sampling - # Excludes: 5-grpc (logprobs=5), 2-None-grpc (n=2 with no logprobs), multiple_choices (n=2 tests) - test_filter: "-k 'not (5-grpc or 2-None-grpc or multiple_choices)'" - setup_vllm: true - ignore_opts: "" - - name: vllm-pd - timeout: 15 - test_dirs: "e2e_test/router/test_pd_mmlu.py" - extra_deps: "" - env_vars: "E2E_RUNTIME=vllm SHOW_WORKER_LOGS=0 SHOW_ROUTER_LOGS=1" - reruns: "--reruns 2 --reruns-delay 5" - parallel_opts: "" - test_filter: "" - setup_vllm: true - ignore_opts: "" - - name: chat-completions-trtllm - timeout: 60 - test_dirs: "e2e_test/chat_completions" - extra_deps: "pytest-parallel py" - env_vars: "E2E_RUNTIME=trtllm SHOW_WORKER_LOGS=1 SHOW_ROUTER_LOGS=1" - reruns: "--reruns 2 --reruns-delay 5" - parallel_opts: "--workers 1 --tests-per-worker 4" - test_filter: "" - setup_trtllm: true - ignore_opts: "" - runner: 4-gpu-h100 + runner: 4-gpu-h100-test + # - name: agentic-apis + # timeout: 30 + # test_dirs: "e2e_test/responses e2e_test/messages" + # extra_deps: "" + # env_vars: "SHOW_WORKER_LOGS=0 SHOW_ROUTER_LOGS=1" + # reruns: "--reruns 2 --reruns-delay 5" + # setup_agentic_deps: true + # parallel_opts: "" + # ignore_opts: "" + # test_filter: "-m 'not external'" + # - name: e2e + # timeout: 20 + # test_dirs: "e2e_test/router e2e_test/embeddings" + # # py is needed for pytest-parallel; sentence-transformers provides HuggingFace reference embeddings + # extra_deps: "pytest-parallel py sentence-transformers" + # env_vars: "SHOW_WORKER_LOGS=0 SHOW_ROUTER_LOGS=1" + # reruns: "--reruns 2 --reruns-delay 5" + # parallel_opts: "--workers 1 --tests-per-worker 4" # Thread-based parallelism + # ignore_opts: "" + # - name: chat-completions-sglang + # timeout: 20 + # test_dirs: "e2e_test/chat_completions" + # extra_deps: "pytest-parallel py" + # env_vars: "E2E_RUNTIME=sglang SHOW_WORKER_LOGS=0 SHOW_ROUTER_LOGS=1" + # reruns: "--reruns 2 --reruns-delay 5" + # parallel_opts: "--workers 1 --tests-per-worker 4" + # test_filter: "" + # ignore_opts: "" + # - name: chat-completions-vllm + # timeout: 20 + # test_dirs: "e2e_test/chat_completions" + # extra_deps: "pytest-parallel py" + # env_vars: "E2E_RUNTIME=vllm SHOW_WORKER_LOGS=0 SHOW_ROUTER_LOGS=1" + # reruns: "--reruns 2 --reruns-delay 5" + # parallel_opts: "--workers 1 --tests-per-worker 4" + # # TODO: Remove filter when vLLM supports logprobs and n>1 with greedy sampling + # # Excludes: 5-grpc (logprobs=5), 2-None-grpc (n=2 with no logprobs), multiple_choices (n=2 tests) + # test_filter: "-k 'not (5-grpc or 2-None-grpc or multiple_choices)'" + # setup_vllm: true + # ignore_opts: "" + # - name: vllm-pd + # timeout: 15 + # test_dirs: "e2e_test/router/test_pd_mmlu.py" + # extra_deps: "" + # env_vars: "E2E_RUNTIME=vllm SHOW_WORKER_LOGS=0 SHOW_ROUTER_LOGS=1" + # reruns: "--reruns 2 --reruns-delay 5" + # parallel_opts: "" + # test_filter: "" + # setup_vllm: true + # ignore_opts: "" + # - name: chat-completions-trtllm + # timeout: 60 + # test_dirs: "e2e_test/chat_completions" + # extra_deps: "pytest-parallel py" + # env_vars: "E2E_RUNTIME=trtllm SHOW_WORKER_LOGS=1 SHOW_ROUTER_LOGS=1" + # reruns: "--reruns 2 --reruns-delay 5" + # parallel_opts: "--workers 1 --tests-per-worker 4" + # test_filter: "" + # setup_trtllm: true + # ignore_opts: "" + # runner: 4-gpu-h100 runs-on: ${{ matrix.runner || 'k8s-runner-gpu' }} timeout-minutes: ${{ matrix.timeout }} steps: diff --git a/scripts/k8s-runner-resources/arc-runner-gpu.yaml b/scripts/k8s-runner-resources/arc-runner-gpu.yaml index 5ca15dc055..8429317050 100644 --- a/scripts/k8s-runner-resources/arc-runner-gpu.yaml +++ b/scripts/k8s-runner-resources/arc-runner-gpu.yaml @@ -81,6 +81,97 @@ spec: --- apiVersion: actions.summerwind.dev/v1alpha1 kind: RunnerDeployment +metadata: + name: arc-runner-gpu-h100-test + namespace: actions-runner-system +spec: + replicas: 1 + template: + spec: + ephemeral: true + repository: lightseekorg/smg + labels: + - 4-gpu-h100-test + serviceAccountName: arc-runner-sa + + nodeSelector: + nvidia.com/gpu: "true" + node.kubernetes.io/instance-type: BM.GPU.H100.8 + + tolerations: + + - key: "nvidia.com/gpu" + operator: "Equal" + value: "true" + effect: "NoSchedule" + + affinity: + podAffinity: + preferredDuringSchedulingIgnoredDuringExecution: + - weight: 100 + podAffinityTerm: + labelSelector: + matchExpressions: + - key: runner-deployment-name + operator: In + values: + - arc-runner-gpu-h100-test + topologyKey: kubernetes.io/hostname + + volumes: + - name: model-cache + persistentVolumeClaim: + claimName: model-cache + - name: docker-sock + emptyDir: {} + - name: docker-storage + emptyDir: + medium: Memory + sizeLimit: 8Gi + - name: dshm + emptyDir: + medium: Memory + sizeLimit: 16Gi + + containers: + - name: runner + image: fra.ocir.io/idqj093njucb/action-runner:v0.0.1 + resources: + limits: + nvidia.com/gpu: 4 + volumeMounts: + - name: model-cache + mountPath: /models + - name: docker-sock + mountPath: /var/run + - name: dshm + mountPath: /dev/shm + env: + - name: DOCKER_HOST + value: unix:///var/run/docker.sock + - name: docker + image: fra.ocir.io/idqj093njucb/docker:dind + securityContext: + privileged: true # Required for DinD + resources: + requests: + cpu: "1" + memory: "2Gi" + limits: + memory: "4Gi" + env: + - name: DOCKER_TLS_CERTDIR + value: "" # Disables TLS for shared socket use + - name: DOCKER_DRIVER + value: overlay2 + volumeMounts: + - name: docker-sock + mountPath: /var/run + - name: docker-storage + mountPath: /var/lib/docker +--- +apiVersion: actions.summerwind.dev/v1alpha1 +kind: RunnerDeployment metadata: name: arc-runner-gpu-a10 namespace: actions-runner-system From 3e2f73c22a79707e63338344c9fa2ec54f64d339 Mon Sep 17 00:00:00 2001 From: key4ng Date: Fri, 27 Feb 2026 00:12:04 +0000 Subject: [PATCH 2/5] chore(ci): refine ignore options in GPU benchmark tests - Removed the ignore option for 'test_pd_perf.py' in the CI workflow, allowing it to run alongside other benchmarks. - Maintained existing configurations for Go and nightly benchmarks. Signed-off-by: key4ng --- .github/workflows/pr-test-rust.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/pr-test-rust.yml b/.github/workflows/pr-test-rust.yml index 3900cb60d2..dc1251351d 100644 --- a/.github/workflows/pr-test-rust.yml +++ b/.github/workflows/pr-test-rust.yml @@ -252,7 +252,7 @@ jobs: reruns: "" upload_benchmarks: true parallel_opts: "" # No parallel for benchmarks (performance measurement) - ignore_opts: "--ignore=e2e_test/benchmarks/test_go_bindings_perf.py --ignore=e2e_test/benchmarks/test_nightly_perf.py --ignore=e2e_test/benchmarks/test_pd_perf.py" # Go and nightly benchmarks run in dedicated jobs + ignore_opts: "--ignore=e2e_test/benchmarks/test_go_bindings_perf.py --ignore=e2e_test/benchmarks/test_nightly_perf.py" # Go and nightly benchmarks run in dedicated jobs runner: 4-gpu-h100-test # - name: agentic-apis # timeout: 30 From 32c476aec7abb37c0dc03f16ec60855e5996bad8 Mon Sep 17 00:00:00 2001 From: key4ng Date: Fri, 27 Feb 2026 00:57:16 +0000 Subject: [PATCH 3/5] chore(ci): enhance GPU runner configuration and restore test setups - Updated the CI workflow to include detailed configurations for various test scenarios, including agentic APIs and chat completions. - Restored previously commented-out test configurations for better clarity and future execution. - Adjusted the runner name back to '4-gpu-h100' for consistency with the updated deployment. Signed-off-by: key4ng --- .github/workflows/pr-test-rust.yml | 124 +++++++++--------- .../k8s-runner-resources/arc-runner-gpu.yaml | 81 ------------ 2 files changed, 62 insertions(+), 143 deletions(-) diff --git a/.github/workflows/pr-test-rust.yml b/.github/workflows/pr-test-rust.yml index dc1251351d..8902b3ac6a 100644 --- a/.github/workflows/pr-test-rust.yml +++ b/.github/workflows/pr-test-rust.yml @@ -253,68 +253,68 @@ jobs: upload_benchmarks: true parallel_opts: "" # No parallel for benchmarks (performance measurement) ignore_opts: "--ignore=e2e_test/benchmarks/test_go_bindings_perf.py --ignore=e2e_test/benchmarks/test_nightly_perf.py" # Go and nightly benchmarks run in dedicated jobs - runner: 4-gpu-h100-test - # - name: agentic-apis - # timeout: 30 - # test_dirs: "e2e_test/responses e2e_test/messages" - # extra_deps: "" - # env_vars: "SHOW_WORKER_LOGS=0 SHOW_ROUTER_LOGS=1" - # reruns: "--reruns 2 --reruns-delay 5" - # setup_agentic_deps: true - # parallel_opts: "" - # ignore_opts: "" - # test_filter: "-m 'not external'" - # - name: e2e - # timeout: 20 - # test_dirs: "e2e_test/router e2e_test/embeddings" - # # py is needed for pytest-parallel; sentence-transformers provides HuggingFace reference embeddings - # extra_deps: "pytest-parallel py sentence-transformers" - # env_vars: "SHOW_WORKER_LOGS=0 SHOW_ROUTER_LOGS=1" - # reruns: "--reruns 2 --reruns-delay 5" - # parallel_opts: "--workers 1 --tests-per-worker 4" # Thread-based parallelism - # ignore_opts: "" - # - name: chat-completions-sglang - # timeout: 20 - # test_dirs: "e2e_test/chat_completions" - # extra_deps: "pytest-parallel py" - # env_vars: "E2E_RUNTIME=sglang SHOW_WORKER_LOGS=0 SHOW_ROUTER_LOGS=1" - # reruns: "--reruns 2 --reruns-delay 5" - # parallel_opts: "--workers 1 --tests-per-worker 4" - # test_filter: "" - # ignore_opts: "" - # - name: chat-completions-vllm - # timeout: 20 - # test_dirs: "e2e_test/chat_completions" - # extra_deps: "pytest-parallel py" - # env_vars: "E2E_RUNTIME=vllm SHOW_WORKER_LOGS=0 SHOW_ROUTER_LOGS=1" - # reruns: "--reruns 2 --reruns-delay 5" - # parallel_opts: "--workers 1 --tests-per-worker 4" - # # TODO: Remove filter when vLLM supports logprobs and n>1 with greedy sampling - # # Excludes: 5-grpc (logprobs=5), 2-None-grpc (n=2 with no logprobs), multiple_choices (n=2 tests) - # test_filter: "-k 'not (5-grpc or 2-None-grpc or multiple_choices)'" - # setup_vllm: true - # ignore_opts: "" - # - name: vllm-pd - # timeout: 15 - # test_dirs: "e2e_test/router/test_pd_mmlu.py" - # extra_deps: "" - # env_vars: "E2E_RUNTIME=vllm SHOW_WORKER_LOGS=0 SHOW_ROUTER_LOGS=1" - # reruns: "--reruns 2 --reruns-delay 5" - # parallel_opts: "" - # test_filter: "" - # setup_vllm: true - # ignore_opts: "" - # - name: chat-completions-trtllm - # timeout: 60 - # test_dirs: "e2e_test/chat_completions" - # extra_deps: "pytest-parallel py" - # env_vars: "E2E_RUNTIME=trtllm SHOW_WORKER_LOGS=1 SHOW_ROUTER_LOGS=1" - # reruns: "--reruns 2 --reruns-delay 5" - # parallel_opts: "--workers 1 --tests-per-worker 4" - # test_filter: "" - # setup_trtllm: true - # ignore_opts: "" - # runner: 4-gpu-h100 + runner: 4-gpu-h100 + - name: agentic-apis + timeout: 30 + test_dirs: "e2e_test/responses e2e_test/messages" + extra_deps: "" + env_vars: "SHOW_WORKER_LOGS=0 SHOW_ROUTER_LOGS=1" + reruns: "--reruns 2 --reruns-delay 5" + setup_agentic_deps: true + parallel_opts: "" + ignore_opts: "" + test_filter: "-m 'not external'" + - name: e2e + timeout: 20 + test_dirs: "e2e_test/router e2e_test/embeddings" + # py is needed for pytest-parallel; sentence-transformers provides HuggingFace reference embeddings + extra_deps: "pytest-parallel py sentence-transformers" + env_vars: "SHOW_WORKER_LOGS=0 SHOW_ROUTER_LOGS=1" + reruns: "--reruns 2 --reruns-delay 5" + parallel_opts: "--workers 1 --tests-per-worker 4" # Thread-based parallelism + ignore_opts: "" + - name: chat-completions-sglang + timeout: 20 + test_dirs: "e2e_test/chat_completions" + extra_deps: "pytest-parallel py" + env_vars: "E2E_RUNTIME=sglang SHOW_WORKER_LOGS=0 SHOW_ROUTER_LOGS=1" + reruns: "--reruns 2 --reruns-delay 5" + parallel_opts: "--workers 1 --tests-per-worker 4" + test_filter: "" + ignore_opts: "" + - name: chat-completions-vllm + timeout: 20 + test_dirs: "e2e_test/chat_completions" + extra_deps: "pytest-parallel py" + env_vars: "E2E_RUNTIME=vllm SHOW_WORKER_LOGS=0 SHOW_ROUTER_LOGS=1" + reruns: "--reruns 2 --reruns-delay 5" + parallel_opts: "--workers 1 --tests-per-worker 4" + # TODO: Remove filter when vLLM supports logprobs and n>1 with greedy sampling + # Excludes: 5-grpc (logprobs=5), 2-None-grpc (n=2 with no logprobs), multiple_choices (n=2 tests) + test_filter: "-k 'not (5-grpc or 2-None-grpc or multiple_choices)'" + setup_vllm: true + ignore_opts: "" + - name: vllm-pd + timeout: 15 + test_dirs: "e2e_test/router/test_pd_mmlu.py" + extra_deps: "" + env_vars: "E2E_RUNTIME=vllm SHOW_WORKER_LOGS=0 SHOW_ROUTER_LOGS=1" + reruns: "--reruns 2 --reruns-delay 5" + parallel_opts: "" + test_filter: "" + setup_vllm: true + ignore_opts: "" + - name: chat-completions-trtllm + timeout: 60 + test_dirs: "e2e_test/chat_completions" + extra_deps: "pytest-parallel py" + env_vars: "E2E_RUNTIME=trtllm SHOW_WORKER_LOGS=1 SHOW_ROUTER_LOGS=1" + reruns: "--reruns 2 --reruns-delay 5" + parallel_opts: "--workers 1 --tests-per-worker 4" + test_filter: "" + setup_trtllm: true + ignore_opts: "" + runner: 4-gpu-h100 runs-on: ${{ matrix.runner || 'k8s-runner-gpu' }} timeout-minutes: ${{ matrix.timeout }} steps: diff --git a/scripts/k8s-runner-resources/arc-runner-gpu.yaml b/scripts/k8s-runner-resources/arc-runner-gpu.yaml index 8429317050..5dc697dddd 100644 --- a/scripts/k8s-runner-resources/arc-runner-gpu.yaml +++ b/scripts/k8s-runner-resources/arc-runner-gpu.yaml @@ -37,87 +37,6 @@ spec: - arc-runner-gpu-h100 topologyKey: kubernetes.io/hostname - volumes: - - name: model-cache - persistentVolumeClaim: - claimName: model-cache - - name: docker-sock - emptyDir: {} - - name: docker-storage - emptyDir: {} - - name: dshm - emptyDir: - medium: Memory - sizeLimit: 16Gi - - containers: - - name: runner - image: fra.ocir.io/idqj093njucb/action-runner:v0.0.1 - resources: - limits: - nvidia.com/gpu: 4 - volumeMounts: - - name: model-cache - mountPath: /models - - name: docker-sock - mountPath: /var/run - - name: dshm - mountPath: /dev/shm - env: - - name: DOCKER_HOST - value: unix:///var/run/docker.sock - - name: docker - image: fra.ocir.io/idqj093njucb/docker:dind - securityContext: - privileged: true # Required for DinD - env: - - name: DOCKER_TLS_CERTDIR - value: "" # Disables TLS for shared socket use - volumeMounts: - - name: docker-sock - mountPath: /var/run - - name: docker-storage - mountPath: /var/lib/docker ---- -apiVersion: actions.summerwind.dev/v1alpha1 -kind: RunnerDeployment -metadata: - name: arc-runner-gpu-h100-test - namespace: actions-runner-system -spec: - replicas: 1 - template: - spec: - ephemeral: true - repository: lightseekorg/smg - labels: - - 4-gpu-h100-test - serviceAccountName: arc-runner-sa - - nodeSelector: - nvidia.com/gpu: "true" - node.kubernetes.io/instance-type: BM.GPU.H100.8 - - tolerations: - - - key: "nvidia.com/gpu" - operator: "Equal" - value: "true" - effect: "NoSchedule" - - affinity: - podAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 100 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: runner-deployment-name - operator: In - values: - - arc-runner-gpu-h100-test - topologyKey: kubernetes.io/hostname - volumes: - name: model-cache persistentVolumeClaim: From 3ad4baf91ca2b79f377f9594bf2a78bfa111fb5e Mon Sep 17 00:00:00 2001 From: key4ng Date: Fri, 27 Feb 2026 01:03:41 +0000 Subject: [PATCH 4/5] chore(ci): update resource limits for GPU runner configuration - Increased CPU limit to 2 for the GPU runner in the Kubernetes setup to enhance performance. - Maintained existing memory limits for consistency. Signed-off-by: key4ng --- scripts/k8s-runner-resources/arc-runner-gpu.yaml | 1 + 1 file changed, 1 insertion(+) diff --git a/scripts/k8s-runner-resources/arc-runner-gpu.yaml b/scripts/k8s-runner-resources/arc-runner-gpu.yaml index 5dc697dddd..66c90639ca 100644 --- a/scripts/k8s-runner-resources/arc-runner-gpu.yaml +++ b/scripts/k8s-runner-resources/arc-runner-gpu.yaml @@ -77,6 +77,7 @@ spec: cpu: "1" memory: "2Gi" limits: + cpu: "2" memory: "4Gi" env: - name: DOCKER_TLS_CERTDIR From 414f55bfa4320b982e3fd624076465fb4285c3e6 Mon Sep 17 00:00:00 2001 From: key4ng Date: Fri, 27 Feb 2026 02:04:55 +0000 Subject: [PATCH 5/5] chore(ci): reduce memory limit for GPU runner configuration - Decreased the memory size limit from 8Gi to 4Gi for the GPU runner in the Kubernetes setup to optimize resource usage. Signed-off-by: key4ng --- scripts/k8s-runner-resources/arc-runner-gpu.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scripts/k8s-runner-resources/arc-runner-gpu.yaml b/scripts/k8s-runner-resources/arc-runner-gpu.yaml index 66c90639ca..3a62d6f3d9 100644 --- a/scripts/k8s-runner-resources/arc-runner-gpu.yaml +++ b/scripts/k8s-runner-resources/arc-runner-gpu.yaml @@ -46,7 +46,7 @@ spec: - name: docker-storage emptyDir: medium: Memory - sizeLimit: 8Gi + sizeLimit: 4Gi - name: dshm emptyDir: medium: Memory