Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 5 additions & 3 deletions .github/workflows/pr-test-rust.yml
Original file line number Diff line number Diff line change
Expand Up @@ -224,6 +224,7 @@ jobs:
upload_benchmarks: true
parallel_opts: "" # No parallel for benchmarks (performance measurement)
ignore_opts: "--ignore=e2e_test/benchmarks/test_go_bindings_perf.py --ignore=e2e_test/benchmarks/test_nightly_perf.py" # Go and nightly benchmarks run in dedicated jobs
runner: 4-gpu-h100
- name: agentic-apis
timeout: 30
test_dirs: "e2e_test/responses e2e_test/messages"
Expand Down Expand Up @@ -284,7 +285,8 @@ jobs:
test_filter: ""
setup_trtllm: true
ignore_opts: ""
runs-on: k8s-runner-gpu
runner: 4-gpu-h100
runs-on: ${{ matrix.runner || 'k8s-runner-gpu' }}
timeout-minutes: ${{ matrix.timeout }}
steps:
- name: Checkout code
Expand All @@ -305,7 +307,7 @@ jobs:
- name: Setup SGLang backend
if: ${{ !matrix.setup_vllm && !matrix.setup_trtllm }}
uses: ./.github/actions/setup-sglang

- name: Download wheel artifact
uses: actions/download-artifact@v7
with:
Expand Down Expand Up @@ -557,7 +559,7 @@ jobs:
if [[ "${{ needs.check-ci.outputs.should_run }}" != "true" ]]; then
echo "CI was skipped (external contributor without run-ci label)"
exit 0
elif [[ "${{ needs.python-lint.result }}" == "failure" || "${{ needs.build-wheel.result }}" == "failure" || "${{ needs.unit-tests.result }}" == "failure" || "${{ needs.gateway-e2e.result }}" == "failure" || "${{ needs.go-unit-tests.result }}" == "failure" || "${{ needs.go-bindings-e2e.result }}" == "failure" ]]; then
elif [[ "${{ needs.python-lint.result }}" == "failure" || "${{ needs.build-wheel.result }}" == "failure" || "${{ needs.python-unit-tests.result }}" == "failure" || "${{ needs.unit-tests.result }}" == "failure" || "${{ needs.gateway-e2e.result }}" == "failure" || "${{ needs.go-unit-tests.result }}" == "failure" || "${{ needs.go-bindings-e2e.result }}" == "failure" ]]; then
echo "One or more jobs failed"
exit 1
else
Expand Down
2 changes: 1 addition & 1 deletion e2e_test/benchmarks/test_pd_perf.py
Original file line number Diff line number Diff line change
Expand Up @@ -20,7 +20,7 @@ def test_pd_perf(self, setup_backend, genai_bench_runner):
# accurate GPU utilization sampling (at least 30+ seconds)
max_requests_per_run=200,
thresholds={
"ttft_mean_max": 13,
"ttft_mean_max": 5,
"e2e_latency_mean_max": 16,
"input_throughput_mean_min": 350,
"output_throughput_mean_min": 18,
Expand Down
2 changes: 1 addition & 1 deletion e2e_test/benchmarks/test_regular_perf.py
Original file line number Diff line number Diff line change
Expand Up @@ -21,7 +21,7 @@ def test_regular_perf(self, setup_backend, genai_bench_runner):
# accurate GPU utilization sampling (at least 30+ seconds)
max_requests_per_run=200,
thresholds={
"ttft_mean_max": 6,
"ttft_mean_max": 0.8,
"e2e_latency_mean_max": 14,
"input_throughput_mean_min": 800,
"output_throughput_mean_min": 12,
Expand Down
39 changes: 39 additions & 0 deletions scripts/k8s-runner-resources/arc-runner-autoscaler.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,39 @@
apiVersion: actions.summerwind.dev/v1alpha1
kind: HorizontalRunnerAutoscaler
metadata:
name: arc-runner-h100-autoscaler
namespace: actions-runner-system
spec:
scaleTargetRef:
kind: RunnerDeployment
name: arc-runner-gpu-h100

minReplicas: 10
maxReplicas: 20

metrics:
- type: PercentageRunnersBusy
scaleUpThreshold: "0.95"
scaleDownThreshold: "0.25"
scaleUpFactor: "0.5"
scaleDownFactor: "0.5"
---
apiVersion: actions.summerwind.dev/v1alpha1
kind: HorizontalRunnerAutoscaler
metadata:
name: arc-cpu-runner-autoscaler
namespace: actions-runner-system
spec:
scaleTargetRef:
kind: RunnerDeployment
name: arc-runner-cpu

minReplicas: 4
maxReplicas: 8

metrics:
- type: PercentageRunnersBusy
scaleUpThreshold: "0.95"
scaleDownThreshold: "0.25"
scaleUpFactor: "0.5"
scaleDownFactor: "0.5"
26 changes: 26 additions & 0 deletions scripts/k8s-runner-resources/arc-runner-cpu.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,26 @@
apiVersion: actions.summerwind.dev/v1alpha1
kind: RunnerDeployment
metadata:
name: arc-runner-cpu
namespace: actions-runner-system
spec:
replicas: 4
template:
spec:
repository: lightseekorg/smg
labels:
- k8s-runner-cpu
serviceAccountName: arc-runner-sa

containers:
- name: runner
image: fra.ocir.io/idqj093njucb/action-runner:v0.0.1
resources:
requests:
cpu: "8"
memory: "16Gi"
limits:
cpu: "8"
memory: "16Gi"
- name: docker
image: fra.ocir.io/idqj093njucb/docker:dind
159 changes: 159 additions & 0 deletions scripts/k8s-runner-resources/arc-runner-gpu.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,159 @@
apiVersion: actions.summerwind.dev/v1alpha1
kind: RunnerDeployment
metadata:
name: arc-runner-gpu-h100
namespace: actions-runner-system
spec:
replicas: 10
template:
spec:
repository: lightseekorg/smg
labels:
- 4-gpu-h100
- k8s-runner-gpu
Comment on lines +12 to +13

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

medium

The podAffinityTerm for arc-runner-gpu-h100 uses runner-deployment-name in its labelSelector. For clarity and explicit control, it would be beneficial to explicitly add this label to the labels section of the runner pod template. This ensures that the affinity rule correctly targets pods belonging to this specific runner deployment.

      labels:
        - 4-gpu-h100
        - k8s-runner-gpu
        - runner-deployment-name: arc-runner-gpu-h100

serviceAccountName: arc-runner-sa

nodeSelector:
nvidia.com/gpu: "true"
beta.kubernetes.io/instance-type: BM.GPU.H100.8

Comment on lines +16 to +19

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

⚠️ Potential issue | 🟠 Major

🧩 Analysis chain

🌐 Web query:

Kubernetes instance-type label beta.kubernetes.io/instance-type deprecated node.kubernetes.io/instance-type

💡 Result:

beta.kubernetes.io/instance-type is a deprecated Node label. Since Kubernetes v1.17, you should use the stable replacement node.kubernetes.io/instance-type instead. [1]

What to change

  • Update any nodeSelector, nodeAffinity, Helm charts, or controllers that match on:
    • beta.kubernetes.io/instance-type=...
  • To instead match on:
    • node.kubernetes.io/instance-type=... [1]

Notes

  • The kubelet typically sets node.kubernetes.io/instance-type based on the cloud provider’s instance type (when a cloud provider integration is in use). [1]
  • Kubernetes generally recommends scheduling on capabilities (CPU/GPU/memory, taints/tolerations, topology) rather than pinning to specific instance types. [1]

Source

  • Kubernetes “Well-Known Labels, Annotations and Taints” reference. [1]

Replace deprecated instance-type label to avoid scheduling failures on newer clusters.

beta.kubernetes.io/instance-type has been deprecated since Kubernetes v1.17. Use the stable replacement node.kubernetes.io/instance-type instead.

🔧 Suggested update
      nodeSelector:
        nvidia.com/gpu: "true"
-       beta.kubernetes.io/instance-type: BM.GPU.H100.8
+       node.kubernetes.io/instance-type: BM.GPU.H100.8
      nodeSelector:
        nvidia.com/gpu: "true"
-       beta.kubernetes.io/instance-type: BM.GPU.A10.4
+       node.kubernetes.io/instance-type: BM.GPU.A10.4

Also applies to: 96-99

🤖 Prompt for AI Agents
In `@scripts/k8s-runner-resources/arc-runner-gpu.yaml` around lines 16 - 19, The
nodeSelector uses the deprecated label key "beta.kubernetes.io/instance-type";
replace that key with the stable "node.kubernetes.io/instance-type" wherever it
appears (e.g., in the nodeSelector block alongside "nvidia.com/gpu: \"true\"")
to ensure scheduling works on newer clusters—update all other occurrences of
"beta.kubernetes.io/instance-type" in the same manifest as well.

tolerations:
- key: "nvidia.com/gpu"
operator: "Equal"
value: "true"
effect: "NoSchedule"

affinity:
podAffinity:
preferredDuringSchedulingIgnoredDuringExecution:
- weight: 100
podAffinityTerm:
labelSelector:
matchExpressions:
- key: runner-deployment-name
operator: In
values:
- arc-runner-gpu-h100
topologyKey: kubernetes.io/hostname

volumes:
- name: model-cache
persistentVolumeClaim:
claimName: model-cache
- name: docker-sock
emptyDir: {}
- name: docker-storage
emptyDir: {}
- name: dshm
emptyDir:
medium: Memory
sizeLimit: 16Gi

containers:
- name: runner
image: fra.ocir.io/idqj093njucb/action-runner:v0.0.1
resources:
limits:
nvidia.com/gpu: 4
volumeMounts:
- name: model-cache
mountPath: /models
- name: docker-sock
mountPath: /var/run
- name: dshm
mountPath: /dev/shm
env:
- name: DOCKER_HOST
value: unix:///var/run/docker.sock
- name: docker
image: fra.ocir.io/idqj093njucb/docker:dind
securityContext:
Comment thread
XinyueZhang369 marked this conversation as resolved.
privileged: true # Required for DinD
env:
- name: DOCKER_TLS_CERTDIR
value: "" # Disables TLS for shared socket use

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

high

Disabling TLS for the Docker socket (DOCKER_TLS_CERTDIR: "") removes a layer of security for communication with the Docker daemon. This is particularly risky when combined with privileged: true as it makes the Docker daemon susceptible to man-in-the-middle attacks if the network is not fully trusted. It's recommended to enable TLS for Docker communication if possible, or ensure the network path is secure.

Copy link
Copy Markdown
Collaborator Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Need this for docker in docker

Comment on lines +69 to +74

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

⚠️ Potential issue | 🟠 Major

🧩 Analysis chain

🏁 Script executed:

# First, let's look at the file to understand the context
wc -l scripts/k8s-runner-resources/arc-runner-gpu.yaml

Repository: lightseekorg/smg

Length of output: 112


🏁 Script executed:

# Read the relevant sections - lines 69-74 and 149-154
sed -n '65,80p' scripts/k8s-runner-resources/arc-runner-gpu.yaml
echo "---"
sed -n '145,160p' scripts/k8s-runner-resources/arc-runner-gpu.yaml

Repository: lightseekorg/smg

Length of output: 1131


🌐 Web query:

Docker-in-Docker DinD Kubernetes security best practices 2025 privileged rootless TLS

💡 Result:

DinD on Kubernetes (2025→2026) security best practices (privileged, rootless, TLS)

1) Prefer not running DinD at all (best risk reduction)

  • Use daemonless / “no Docker socket” builders where possible (e.g., BuildKit rootless or Buildx Kubernetes driver with rootless=true) to avoid giving a pod a Docker daemon with elevated host-like power. Docker’s Buildx Kubernetes driver explicitly supports rootless mode and creates pods without securityContext.privileged. [1]

2) If you must run DinD, treat it as high risk

  • Assume privileged == near-host-level power: Kubernetes notes privileged containers override/ignore key hardening controls (seccomp/AppArmor/SELinux) and effectively gain broad capabilities. Avoid unless there’s no alternative. [2]
  • Enforce Pod Security Standards:
    • For most namespaces, enforce Baseline or Restricted so privileged: true, host namespaces, and hostPath mounts are disallowed by policy. [3]
    • Put the DinD workload in a tightly controlled exception namespace with explicit approvals.

3) Rootless: helpful, but don’t assume it removes the need for privilege

  • Rootless Docker runs the daemon and containers as a non-root user via user namespaces (reduces impact of daemon/runtime bugs). [4]
  • However, Docker’s own guidance for “Rootless Docker in Docker” still uses --privileged (to disable seccomp/AppArmor and mount restrictions), meaning you may still be bypassing important isolation controls even if the daemon UID isn’t 0. [5]

4) Use Kubernetes User Namespaces where available (additional containment)

  • Consider opting pods into user namespaces with spec.hostUsers: false so “root in the container” maps to an unprivileged host UID/GID, reducing host impact of a breakout. [6]
  • Note: user namespaces come with constraints/limitations (for example, cannot combine with hostNetwork/hostPID/hostIPC). [6]

5) Never expose an unauthenticated Docker API; use TLS or SSH

  • If you expose the Docker daemon beyond a local UNIX socket, Docker recommends securing it with SSH or mutual TLS (tlsverify). [7]

6) If privileged DinD is unavoidable, reduce blast radius aggressively

  • Dedicate a separate node pool (taints/tolerations) for DinD jobs; assume node compromise is possible.
  • Use tight NetworkPolicies (only allow what’s required) and short-lived, ephemeral workers.
  • Minimize mounts: avoid hostPath; don’t mount the host’s /var/run/docker.sock (that is effectively root on the node in many setups).
  • Keep privileged as a last resort; otherwise, grant only specific Linux capabilities (Kubernetes recommends capabilities over privileged when possible). [2]

Key takeaway: In Kubernetes, “DinD” usually pushes you toward privileged (or “privileged-like”) operation; the safest pattern is to avoid DinD and use rootless BuildKit/Buildx-on-Kubernetes where possible. [1][2][5]

Sources: Docker docs on Buildx Kubernetes rootless mode [1], Kubernetes on privileged container impacts [2], Kubernetes Pod Security Standards [3], Docker rootless mode [4], Docker rootless DinD tip showing --privileged [5], Kubernetes user namespaces (hostUsers: false) [6], Docker daemon TLS/SSH guidance [7].


Privileged DinD with TLS disabled is a security risk for CI runners.

A privileged Docker daemon plus DOCKER_TLS_CERTDIR="" gives any job container full daemon control. For untrusted PRs, this can lead to host escape. Consider rootless Docker, a non-privileged runtime (containerd/buildkit), or enabling TLS with scoped credentials.

This applies to both locations: lines 69-74 and 149-154.

🤖 Prompt for AI Agents
In `@scripts/k8s-runner-resources/arc-runner-gpu.yaml` around lines 69 - 74, The
deployment uses privileged DinD (image: fra.ocir.io/idqj093njucb/docker:dind)
with securityContext.privileged: true and DOCKER_TLS_CERTDIR set to "", which
allows job containers full daemon control; replace this by either running
rootless Docker or a non-privileged build runtime (e.g., switch to
containerd/buildkit-based image and remove securityContext.privileged), or
enable TLS by removing DOCKER_TLS_CERTDIR:"" and configuring DOCKER_TLS_CERTDIR
to a secure path plus mounting scoped TLS credentials/secrets for the runner;
apply the same change for both DinD blocks that set image: fra.ocir.io/.../dind,
securityContext.privileged and env DOCKER_TLS_CERTDIR.

volumeMounts:
- name: docker-sock
mountPath: /var/run
- name: docker-storage
mountPath: /var/lib/docker
---
apiVersion: actions.summerwind.dev/v1alpha1
kind: RunnerDeployment
metadata:
name: arc-runner-gpu-a10
namespace: actions-runner-system
spec:
replicas: 2

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I thought we have 4 A10, do we only have 2? I also didn't see hpa for a10.

@XinyueZhang369 XinyueZhang369 Feb 7, 2026 •

Copy link
Copy Markdown
Collaborator Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

we probably don't need to provision all 4 A20 all the time, so here is set to 2 for now. For HPA, today the auto scaler somehow kept creating cpu and a10 runner pods that cannot register to the repo regardless the max number, so I deleted all the old auto scalers for cpu, a10 and h100, this one is a new configuration, I want to bake it for some times, since most resources are h100, I only create for h100 for now for baking, once the scaling strategy works stably, I'll create the same for a10 and update this file

template:
spec:
repository: lightseekorg/smg
labels:
- 4-gpu-a10
- k8s-runner-gpu
Comment on lines +92 to +93

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

medium

Similar to the H100 runner, the podAffinityTerm for arc-runner-gpu-a10 uses runner-deployment-name in its labelSelector. Please explicitly add this label to the labels section of the runner pod template for clarity and to ensure the affinity rule correctly targets pods belonging to this specific runner deployment.

      labels:
        - 4-gpu-a10
        - k8s-runner-gpu
        - runner-deployment-name: arc-runner-gpu-a10

serviceAccountName: arc-runner-sa

nodeSelector:
nvidia.com/gpu: "true"
beta.kubernetes.io/instance-type: BM.GPU.A10.4

tolerations:
- key: "nvidia.com/gpu"
operator: "Equal"
value: "true"
effect: "NoSchedule"

affinity:
podAffinity:
preferredDuringSchedulingIgnoredDuringExecution:
- weight: 100
podAffinityTerm:
labelSelector:
matchExpressions:
- key: runner-deployment-name
operator: In
values:
- arc-runner-gpu-a10
topologyKey: kubernetes.io/hostname

volumes:
- name: model-cache
persistentVolumeClaim:
claimName: model-cache
- name: docker-sock
emptyDir: {}
- name: docker-storage
emptyDir: {}
- name: dshm
emptyDir:
medium: Memory
sizeLimit: 16Gi

containers:
- name: runner
image: fra.ocir.io/idqj093njucb/action-runner:v0.0.1
resources:
limits:
nvidia.com/gpu: 4
volumeMounts:
- name: model-cache
mountPath: /models
- name: docker-sock
mountPath: /var/run
- name: dshm
mountPath: /dev/shm
env:
- name: DOCKER_HOST
value: unix:///var/run/docker.sock
- name: docker
image: fra.ocir.io/idqj093njucb/docker:dind
securityContext:
privileged: true # Required for DinD
env:
- name: DOCKER_TLS_CERTDIR
value: "" # Disables TLS for shared socket use
volumeMounts:
- name: docker-sock
mountPath: /var/run
- name: docker-storage
mountPath: /var/lib/docker
45 changes: 45 additions & 0 deletions scripts/k8s-runner-resources/arc-runner-rbac.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,45 @@
apiVersion: v1
kind: ServiceAccount
metadata:
name: arc-runner-sa
namespace: actions-runner-system
---
apiVersion: rbac.authorization.k8s.io/v1
kind: Role
metadata:
name: arc-runner
namespace: actions-runner-system
rules:
# Argo Workflows
- apiGroups: [""]
resources:
- secrets
verbs:
- get
- list
- watch

# Pods
- apiGroups: [""]
resources:
- pods
- pods/log
- pods/exec
verbs:
Comment thread
XinyueZhang369 marked this conversation as resolved.
- get
- list
- watch
Comment thread
XinyueZhang369 marked this conversation as resolved.
---
apiVersion: rbac.authorization.k8s.io/v1
kind: RoleBinding
metadata:
name: arc-runner-rb
namespace: actions-runner-system
roleRef:
apiGroup: rbac.authorization.k8s.io
kind: Role
name: arc-runner
subjects:
- kind: ServiceAccount
name: arc-runner-sa
namespace: actions-runner-system