From b1613bb7b3c009e61a667093177ae624bfc1fe10 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 6 Mar 2026 09:50:31 +0000 Subject: [PATCH 1/9] Speed up regression evals CI pipeline - Add pip/poetry dependency caching to reduce Python install time - Run KIND cluster setup in background, parallel with Holmes env setup (saves ~70s by overlapping the two independent setup phases) - Install kube-prometheus-stack and metrics-server in parallel within KIND setup (saves ~26s) - Reduce Calico CNI wait loop intervals from 5s to 2s - Remove unnecessary test pod verification step from cluster readiness - Scale pytest parallelism dynamically based on test count (6-20 workers) https://claude.ai/code/session_01ReC3SkRaWTNe3BZyuhLRyx Signed-off-by: Claude --- .github/actions/setup-holmes-env/action.yml | 13 +- .github/workflows/eval-regression.yaml | 176 +++++++++++++++++++- 2 files changed, 181 insertions(+), 8 deletions(-) diff --git a/.github/actions/setup-holmes-env/action.yml b/.github/actions/setup-holmes-env/action.yml index c3be7f66bf..960fbb6eed 100644 --- a/.github/actions/setup-holmes-env/action.yml +++ b/.github/actions/setup-holmes-env/action.yml @@ -23,7 +23,7 @@ runs: with: python-version: ${{ inputs.python-version }} - - name: Install Python dependencies and Poetry + - name: Install Poetry shell: bash run: | python -m pip install --upgrade pip setuptools pyinstaller @@ -37,6 +37,17 @@ runs: # Configure Poetry to not create virtualenvs (we use system Python in CI) $HOME/.local/bin/poetry config virtualenvs.create false + - name: Cache Python dependencies + uses: actions/cache@v4 + with: + path: | + ~/.cache/pip + ~/.cache/pypoetry + # hashFiles is relative to GITHUB_WORKSPACE + key: poetry-deps-${{ inputs.python-version }}-${{ hashFiles(format('{0}/poetry.lock', inputs.working-directory)) }} + restore-keys: | + poetry-deps-${{ inputs.python-version }}- + - name: Install project dependencies shell: bash working-directory: ${{ inputs.working-directory }} diff --git a/.github/workflows/eval-regression.yaml b/.github/workflows/eval-regression.yaml index f15dac310e..00441866ae 100644 --- a/.github/workflows/eval-regression.yaml +++ b/.github/workflows/eval-regression.yaml @@ -594,7 +594,143 @@ jobs: core.setOutput('is_persistent', 'false'); } - # Reordered: Setup HolmesGPT first (needed for pytest), then collect evals, then KIND + # Write KIND setup script to a file, then run it in background + # This runs in parallel with HolmesGPT environment setup, saving ~70s + - name: Start KIND cluster setup (background) + if: steps.check-tests.outputs.should-run == 'true' + shell: bash + run: | + # Write the setup script to a file to avoid heredoc/indentation issues + cat > /tmp/kind-setup.sh << 'SCRIPT_EOF' + #!/bin/bash + set -e + + # Install KIND + curl -Lo ./kind https://kind.sigs.k8s.io/dl/v0.31.0/kind-linux-amd64 + chmod +x ./kind + sudo mv ./kind /usr/local/bin/kind + kind version + + # Create KIND cluster config + cat > /tmp/kind-config.yaml << 'EOF' + kind: Cluster + apiVersion: kind.x-k8s.io/v1alpha4 + networking: + disableDefaultCNI: true + nodes: + - role: control-plane + extraPortMappings: + - containerPort: 30000 + hostPort: 30000 + protocol: TCP + - containerPort: 30001 + hostPort: 30001 + protocol: TCP + kubeadmConfigPatches: + - | + kind: InitConfiguration + nodeRegistration: + kubeletExtraArgs: + max-pods: "300" + - | + kind: KubeProxyConfiguration + metricsBindAddress: "0.0.0.0:10249" + - | + kind: KubeletConfiguration + maxPods: 300 + EOF + + kind create cluster --name kind --config /tmp/kind-config.yaml + kubectl cluster-info --context kind-kind + + # Install Calico CNI (reduced wait intervals: 2s instead of 5s) + CALICO_VERSION="v3.31.3" + kubectl create -f "https://raw.githubusercontent.com/projectcalico/calico/${CALICO_VERSION}/manifests/tigera-operator.yaml" + + for i in $(seq 1 90); do + kubectl get pods -n tigera-operator --no-headers 2>/dev/null | grep -q "tigera-operator" && break + sleep 2 + done + kubectl wait --for=condition=Ready --timeout=300s -n tigera-operator pods --all + + for i in $(seq 1 90); do + kubectl get crd installations.operator.tigera.io 2>/dev/null && break + sleep 2 + done + kubectl wait --for=condition=Established --timeout=120s crd/installations.operator.tigera.io + + # Create Calico installation + cat << 'EOF' | kubectl create -f - + apiVersion: operator.tigera.io/v1 + kind: Installation + metadata: + name: default + spec: + calicoNetwork: + ipPools: + - blockSize: 26 + cidr: 10.244.0.0/16 + encapsulation: IPIP + natOutgoing: Enabled + nodeSelector: all() + EOF + + for i in $(seq 1 90); do + kubectl get pods -n calico-system --no-headers 2>/dev/null | grep -q "calico" && break + sleep 2 + done + kubectl wait --for=condition=Ready --timeout=300s -n calico-system pods --all + + kubectl wait --for=condition=Ready nodes --all --timeout=300s + kubectl wait --for=condition=Ready pods --all -n kube-system --timeout=300s + + # Install Helm + curl https://raw.githubusercontent.com/helm/helm/main/scripts/get-helm-3 | bash + + # Install kube-prometheus-stack and metrics-server in parallel + ( + helm repo add prometheus-community https://prometheus-community.github.io/helm-charts + helm repo update + helm install robusta prometheus-community/kube-prometheus-stack \ + --version 72.6.2 \ + --namespace default \ + --set prometheus.prometheusSpec.podMonitorSelectorNilUsesHelmValues=false \ + --set prometheus.prometheusSpec.serviceMonitorSelectorNilUsesHelmValues=false \ + --set prometheus.prometheusSpec.scrapeInterval=15s \ + --set prometheus.prometheusSpec.resources.requests.memory=256Mi \ + --set prometheus.prometheusSpec.resources.requests.cpu=100m \ + --set prometheus.prometheusSpec.resources.limits.memory=512Mi \ + --set prometheus.prometheusSpec.resources.limits.cpu=500m \ + --set alertmanager.enabled=false \ + --set grafana.enabled=false \ + --set kubeStateMetrics.enabled=true \ + --set nodeExporter.enabled=true \ + --wait \ + --timeout 5m + kubectl wait --for=condition=Ready pods -l app.kubernetes.io/name=prometheus -n default --timeout=300s + ) & + PROM_PID=$! + + ( + kubectl apply -f https://github.com/kubernetes-sigs/metrics-server/releases/download/v0.7.2/components.yaml + kubectl patch deployment metrics-server -n kube-system \ + --type='json' \ + -p='[{"op": "add", "path": "/spec/template/spec/containers/0/args/-", "value": "--kubelet-insecure-tls"}]' + kubectl wait --for=condition=Available deployment/metrics-server -n kube-system --timeout=120s + ) & + METRICS_PID=$! + + wait $PROM_PID + wait $METRICS_PID + echo "KIND_SETUP_COMPLETE" + SCRIPT_EOF + + chmod +x /tmp/kind-setup.sh + # Run in background, log to file + /tmp/kind-setup.sh > /tmp/kind-setup.log 2>&1 & + echo $! > /tmp/kind-setup-pid + echo "KIND cluster setup started in background (PID: $(cat /tmp/kind-setup-pid))" + - name: Setup HolmesGPT environment if: steps.check-tests.outputs.should-run == 'true' uses: ./code/.github/actions/setup-holmes-env @@ -741,12 +877,31 @@ jobs: body: body }); - - name: Setup KIND cluster + - name: Wait for KIND cluster setup if: steps.check-tests.outputs.should-run == 'true' - uses: ./code/.github/actions/setup-kind-cluster - with: - cluster-name: 'kind' - wait-for-ready: 'true' + shell: bash + run: | + echo "Waiting for background KIND cluster setup to complete..." + KIND_PID=$(cat /tmp/kind-setup-pid) + # Poll until the background process finishes (can't use wait across shell invocations) + while kill -0 "$KIND_PID" 2>/dev/null; do + sleep 5 + done + # Check exit status via the log sentinel + if grep -q "KIND_SETUP_COMPLETE" /tmp/kind-setup.log; then + echo "KIND cluster setup completed successfully" + else + echo "KIND cluster setup FAILED. Full logs:" + cat /tmp/kind-setup.log + exit 1 + fi + # Show summary + echo "=== KIND setup log (last 20 lines) ===" + tail -20 /tmp/kind-setup.log + echo "=== Cluster status ===" + kubectl get nodes + kubectl get pods -n kube-system --no-headers | head -5 + kubectl get pods -n calico-system --no-headers - name: Update progress - KIND ready, starting evals if: steps.check-tests.outputs.should-run == 'true' && steps.eval-params.outputs.pr_number != '' && steps.initial-comment.outputs.comment_id != '' @@ -849,7 +1004,14 @@ jobs: START_TIME=$(date +%s) - PYTEST_ARGS=(--no-cov tests/llm/test_ask_holmes.py tests/llm/test_investigate.py -s -n10 -m "$EVAL_MARKER_EXPR") + # Scale parallelism with test count (min 6, max 20, capped at nproc) + TEST_COUNT=${{ steps.test-preview.outputs.test_count || '10' }} + NPROC=$(nproc) + N_WORKERS=$((TEST_COUNT < 6 ? 6 : (TEST_COUNT > 20 ? 20 : TEST_COUNT))) + N_WORKERS=$((N_WORKERS > NPROC ? NPROC : N_WORKERS)) + echo "Running $TEST_COUNT tests with $N_WORKERS parallel workers" + + PYTEST_ARGS=(--no-cov tests/llm/test_ask_holmes.py tests/llm/test_investigate.py -s -n"$N_WORKERS" -m "$EVAL_MARKER_EXPR") [[ -n "$EVAL_FILTER" ]] && PYTEST_ARGS+=(-k "$EVAL_FILTER") poetry run pytest "${PYTEST_ARGS[@]}" || true From 14b9cb3fad1b177711721f976429574a37fe453d Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 6 Mar 2026 10:11:13 +0000 Subject: [PATCH 2/9] Use fixed 20 parallel workers for evals (IO-bound, not CPU-bound) https://claude.ai/code/session_01ReC3SkRaWTNe3BZyuhLRyx Signed-off-by: Claude --- .github/workflows/eval-regression.yaml | 9 +-------- 1 file changed, 1 insertion(+), 8 deletions(-) diff --git a/.github/workflows/eval-regression.yaml b/.github/workflows/eval-regression.yaml index 00441866ae..dfc2598723 100644 --- a/.github/workflows/eval-regression.yaml +++ b/.github/workflows/eval-regression.yaml @@ -1004,14 +1004,7 @@ jobs: START_TIME=$(date +%s) - # Scale parallelism with test count (min 6, max 20, capped at nproc) - TEST_COUNT=${{ steps.test-preview.outputs.test_count || '10' }} - NPROC=$(nproc) - N_WORKERS=$((TEST_COUNT < 6 ? 6 : (TEST_COUNT > 20 ? 20 : TEST_COUNT))) - N_WORKERS=$((N_WORKERS > NPROC ? NPROC : N_WORKERS)) - echo "Running $TEST_COUNT tests with $N_WORKERS parallel workers" - - PYTEST_ARGS=(--no-cov tests/llm/test_ask_holmes.py tests/llm/test_investigate.py -s -n"$N_WORKERS" -m "$EVAL_MARKER_EXPR") + PYTEST_ARGS=(--no-cov tests/llm/test_ask_holmes.py tests/llm/test_investigate.py -s -n20 -m "$EVAL_MARKER_EXPR") [[ -n "$EVAL_FILTER" ]] && PYTEST_ARGS+=(-k "$EVAL_FILTER") poetry run pytest "${PYTEST_ARGS[@]}" || true From 79f18cd2b49e554a2499bbd8b4e4bddb614a636e Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 6 Mar 2026 10:19:33 +0000 Subject: [PATCH 3/9] Trigger CI rerun to test dependency caching https://claude.ai/code/session_01ReC3SkRaWTNe3BZyuhLRyx Signed-off-by: Claude --- .github/workflows/eval-regression.yaml | 1 + 1 file changed, 1 insertion(+) diff --git a/.github/workflows/eval-regression.yaml b/.github/workflows/eval-regression.yaml index dfc2598723..a08450463c 100644 --- a/.github/workflows/eval-regression.yaml +++ b/.github/workflows/eval-regression.yaml @@ -1135,3 +1135,4 @@ jobs: else echo "✅ All tests passed." fi +# CI speed optimizations From 569a4376d5dbe348f6e4b686369024115a9fb47d Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 6 Mar 2026 10:24:19 +0000 Subject: [PATCH 4/9] Fix SIGPIPE (exit 141) in KIND wait step from head piping https://claude.ai/code/session_01ReC3SkRaWTNe3BZyuhLRyx Signed-off-by: Claude --- .github/workflows/eval-regression.yaml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/.github/workflows/eval-regression.yaml b/.github/workflows/eval-regression.yaml index a08450463c..5d8a971e7a 100644 --- a/.github/workflows/eval-regression.yaml +++ b/.github/workflows/eval-regression.yaml @@ -900,8 +900,8 @@ jobs: tail -20 /tmp/kind-setup.log echo "=== Cluster status ===" kubectl get nodes - kubectl get pods -n kube-system --no-headers | head -5 - kubectl get pods -n calico-system --no-headers + kubectl get pods -n kube-system --no-headers | head -5 || true + kubectl get pods -n calico-system --no-headers || true - name: Update progress - KIND ready, starting evals if: steps.check-tests.outputs.should-run == 'true' && steps.eval-params.outputs.pr_number != '' && steps.initial-comment.outputs.comment_id != '' From e09f672bebcb65711e4433ebdc963d5edad1d0ea Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 6 Mar 2026 12:06:27 +0000 Subject: [PATCH 5/9] Add caching for KIND cluster infrastructure setup Cache KIND binary, Helm binary, KIND node Docker image, Calico/metrics-server manifests, Helm chart repository data, and all workload container images (Calico, Prometheus, kube-state-metrics, node-exporter, metrics-server). On cache hit: - Binaries loaded from cache instead of downloading (~10s saved) - KIND node image loaded via docker load (~15-20s saved) - Workload images pre-loaded into KIND via kind load image-archive, so pods start without pulling from registries (~40-60s saved) - Helm charts served from local cache (~10s saved) - Manifests read from local files (~5s saved) On cache miss (first run): artifacts are saved for subsequent runs. Cache key includes all component versions to auto-bust on upgrades. https://claude.ai/code/session_01ReC3SkRaWTNe3BZyuhLRyx Signed-off-by: Claude --- .github/workflows/eval-regression.yaml | 108 ++++++++++++++++++++++--- 1 file changed, 96 insertions(+), 12 deletions(-) diff --git a/.github/workflows/eval-regression.yaml b/.github/workflows/eval-regression.yaml index 5d8a971e7a..581cc91ee8 100644 --- a/.github/workflows/eval-regression.yaml +++ b/.github/workflows/eval-regression.yaml @@ -594,8 +594,18 @@ jobs: core.setOutput('is_persistent', 'false'); } + # Cache KIND infrastructure: binaries, Docker images, Helm charts, manifests + - name: Cache KIND infrastructure + if: steps.check-tests.outputs.should-run == 'true' + uses: actions/cache@v4 + id: kind-cache + with: + path: ~/kind-cache + key: kind-infra-v1-kind0.31.0-calico3.31.3-prom72.6.2-metrics0.7.2 + # Write KIND setup script to a file, then run it in background # This runs in parallel with HolmesGPT environment setup, saving ~70s + # Caching saves an additional ~60s on subsequent runs - name: Start KIND cluster setup (background) if: steps.check-tests.outputs.should-run == 'true' shell: bash @@ -605,13 +615,34 @@ jobs: #!/bin/bash set -e - # Install KIND - curl -Lo ./kind https://kind.sigs.k8s.io/dl/v0.31.0/kind-linux-amd64 - chmod +x ./kind - sudo mv ./kind /usr/local/bin/kind + CACHE_DIR="$HOME/kind-cache" + mkdir -p "$CACHE_DIR/bin" "$CACHE_DIR/manifests" "$CACHE_DIR/images" "$CACHE_DIR/helm-cache" "$CACHE_DIR/helm-data" + + # Use cached Helm repository and chart data + export HELM_CACHE_HOME="$CACHE_DIR/helm-cache" + export HELM_DATA_HOME="$CACHE_DIR/helm-data" + + # === Install KIND binary (cached or download) === + if [ -f "$CACHE_DIR/bin/kind" ]; then + echo "Using cached KIND binary" + sudo cp "$CACHE_DIR/bin/kind" /usr/local/bin/kind + sudo chmod +x /usr/local/bin/kind + else + echo "Downloading KIND binary" + curl -sLo "$CACHE_DIR/bin/kind" https://kind.sigs.k8s.io/dl/v0.31.0/kind-linux-amd64 + chmod +x "$CACHE_DIR/bin/kind" + sudo cp "$CACHE_DIR/bin/kind" /usr/local/bin/kind + fi kind version - # Create KIND cluster config + # === Load cached KIND node image if available === + KIND_NODE_IMAGE="kindest/node:v1.35.0" + if [ -f "$CACHE_DIR/images/kind-node.tar" ]; then + echo "Loading cached KIND node image" + docker load -i "$CACHE_DIR/images/kind-node.tar" + fi + + # === Create KIND cluster === cat > /tmp/kind-config.yaml << 'EOF' kind: Cluster apiVersion: kind.x-k8s.io/v1alpha4 @@ -640,12 +671,31 @@ jobs: maxPods: 300 EOF - kind create cluster --name kind --config /tmp/kind-config.yaml + kind create cluster --name kind --config /tmp/kind-config.yaml --image "$KIND_NODE_IMAGE" kubectl cluster-info --context kind-kind - # Install Calico CNI (reduced wait intervals: 2s instead of 5s) + # Save KIND node image in background for cache (first run only) + if [ ! -f "$CACHE_DIR/images/kind-node.tar" ]; then + echo "Saving KIND node image for cache" + docker save "$KIND_NODE_IMAGE" -o "$CACHE_DIR/images/kind-node.tar" & + SAVE_NODE_PID=$! + fi + + # === Load cached workload images into KIND (Calico, Prometheus, etc.) === + if [ -f "$CACHE_DIR/images/workload-images.tar" ]; then + echo "Loading cached workload images into KIND" + kind load image-archive "$CACHE_DIR/images/workload-images.tar" --name kind + echo "Cached workload images loaded" + fi + + # === Install Calico CNI === CALICO_VERSION="v3.31.3" - kubectl create -f "https://raw.githubusercontent.com/projectcalico/calico/${CALICO_VERSION}/manifests/tigera-operator.yaml" + CALICO_MANIFEST="$CACHE_DIR/manifests/calico-operator-${CALICO_VERSION}.yaml" + if [ ! -f "$CALICO_MANIFEST" ]; then + echo "Downloading Calico operator manifest" + curl -sLo "$CALICO_MANIFEST" "https://raw.githubusercontent.com/projectcalico/calico/${CALICO_VERSION}/manifests/tigera-operator.yaml" + fi + kubectl create -f "$CALICO_MANIFEST" for i in $(seq 1 90); do kubectl get pods -n tigera-operator --no-headers 2>/dev/null | grep -q "tigera-operator" && break @@ -684,10 +734,24 @@ jobs: kubectl wait --for=condition=Ready nodes --all --timeout=300s kubectl wait --for=condition=Ready pods --all -n kube-system --timeout=300s - # Install Helm - curl https://raw.githubusercontent.com/helm/helm/main/scripts/get-helm-3 | bash + # === Install Helm (cached or download) === + if [ -f "$CACHE_DIR/bin/helm" ]; then + echo "Using cached Helm binary" + sudo cp "$CACHE_DIR/bin/helm" /usr/local/bin/helm + sudo chmod +x /usr/local/bin/helm + else + echo "Downloading Helm" + curl -sL https://raw.githubusercontent.com/helm/helm/main/scripts/get-helm-3 | bash + cp "$(which helm)" "$CACHE_DIR/bin/helm" + fi + + # === Install kube-prometheus-stack and metrics-server in parallel === + METRICS_MANIFEST="$CACHE_DIR/manifests/metrics-server-v0.7.2.yaml" + if [ ! -f "$METRICS_MANIFEST" ]; then + echo "Downloading metrics-server manifest" + curl -sLo "$METRICS_MANIFEST" "https://github.com/kubernetes-sigs/metrics-server/releases/download/v0.7.2/components.yaml" + fi - # Install kube-prometheus-stack and metrics-server in parallel ( helm repo add prometheus-community https://prometheus-community.github.io/helm-charts helm repo update @@ -712,7 +776,7 @@ jobs: PROM_PID=$! ( - kubectl apply -f https://github.com/kubernetes-sigs/metrics-server/releases/download/v0.7.2/components.yaml + kubectl apply -f "$METRICS_MANIFEST" kubectl patch deployment metrics-server -n kube-system \ --type='json' \ -p='[{"op": "add", "path": "/spec/template/spec/containers/0/args/-", "value": "--kubelet-insecure-tls"}]' @@ -722,6 +786,26 @@ jobs: wait $PROM_PID wait $METRICS_PID + + # === Save workload images from KIND for cache (first run only) === + if [ ! -f "$CACHE_DIR/images/workload-images.tar" ]; then + echo "Saving workload images for cache (first run only)" + IMAGES=$(docker exec kind-control-plane crictl images -o json \ + | jq -r '.images[].repoTags[]' \ + | grep -v '' \ + | grep -v 'kindest' \ + | sort) + if [ -n "$IMAGES" ]; then + docker exec kind-control-plane ctr -n k8s.io images export /tmp/workload-images.tar $IMAGES + docker cp kind-control-plane:/tmp/workload-images.tar "$CACHE_DIR/images/workload-images.tar" + docker exec kind-control-plane rm -f /tmp/workload-images.tar + echo "Saved $(echo "$IMAGES" | wc -l) workload images for cache" + fi + fi + + # Wait for background KIND node image save + [ -n "${SAVE_NODE_PID:-}" ] && wait $SAVE_NODE_PID || true + echo "KIND_SETUP_COMPLETE" SCRIPT_EOF From c26ec28afc373f151feafd3bf4df537565ecf6a4 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 6 Mar 2026 12:23:15 +0000 Subject: [PATCH 6/9] Cache Python virtualenv instead of just pip/poetry download caches The previous cache only stored ~/.cache/pip and ~/.cache/pypoetry (download caches). With virtualenvs.create=false, poetry install still had to unpack and install every package to system site-packages on every run. Now uses in-project virtualenv (.venv/) and caches it directly. On cache hit, poetry install is near-instant since all packages are already installed. Also adds .venv/bin to PATH so scripts with #!/usr/bin/env python3 work. https://claude.ai/code/session_01ReC3SkRaWTNe3BZyuhLRyx Signed-off-by: Claude --- .github/actions/setup-holmes-env/action.yml | 21 +++++++++++---------- 1 file changed, 11 insertions(+), 10 deletions(-) diff --git a/.github/actions/setup-holmes-env/action.yml b/.github/actions/setup-holmes-env/action.yml index 960fbb6eed..28ce4977f1 100644 --- a/.github/actions/setup-holmes-env/action.yml +++ b/.github/actions/setup-holmes-env/action.yml @@ -34,19 +34,14 @@ runs: # Add Poetry to PATH echo "$HOME/.local/bin" >> $GITHUB_PATH - # Configure Poetry to not create virtualenvs (we use system Python in CI) - $HOME/.local/bin/poetry config virtualenvs.create false + # Use in-project virtualenv for efficient caching + $HOME/.local/bin/poetry config virtualenvs.in-project true - - name: Cache Python dependencies + - name: Cache Python virtualenv uses: actions/cache@v4 with: - path: | - ~/.cache/pip - ~/.cache/pypoetry - # hashFiles is relative to GITHUB_WORKSPACE - key: poetry-deps-${{ inputs.python-version }}-${{ hashFiles(format('{0}/poetry.lock', inputs.working-directory)) }} - restore-keys: | - poetry-deps-${{ inputs.python-version }}- + path: ${{ inputs.working-directory }}/.venv + key: venv-${{ inputs.python-version }}-${{ hashFiles(format('{0}/poetry.lock', inputs.working-directory)) }} - name: Install project dependencies shell: bash @@ -54,6 +49,12 @@ runs: run: | poetry install --no-root --with dev + - name: Add virtualenv to PATH + shell: bash + working-directory: ${{ inputs.working-directory }} + run: | + echo "$(pwd)/.venv/bin" >> $GITHUB_PATH + - name: Install kubectl if: inputs.install-kubectl == 'true' shell: bash From 342ccd198aa182254c4dafea1f387b243d92a5fc Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 6 Mar 2026 12:39:43 +0000 Subject: [PATCH 7/9] Fix ctr images export failure on multi-arch manifest lists ctr images export fails with "content digest not found" when containerd has multi-arch manifest lists but only the linux/amd64 layers are present. Fix by using --platform linux/amd64 flag, with fallback to per-image export if bulk export still fails. https://claude.ai/code/session_01ReC3SkRaWTNe3BZyuhLRyx Signed-off-by: Claude --- .github/workflows/eval-regression.yaml | 33 ++++++++++++++++++++++---- 1 file changed, 29 insertions(+), 4 deletions(-) diff --git a/.github/workflows/eval-regression.yaml b/.github/workflows/eval-regression.yaml index 581cc91ee8..732a41996f 100644 --- a/.github/workflows/eval-regression.yaml +++ b/.github/workflows/eval-regression.yaml @@ -796,10 +796,35 @@ jobs: | grep -v 'kindest' \ | sort) if [ -n "$IMAGES" ]; then - docker exec kind-control-plane ctr -n k8s.io images export /tmp/workload-images.tar $IMAGES - docker cp kind-control-plane:/tmp/workload-images.tar "$CACHE_DIR/images/workload-images.tar" - docker exec kind-control-plane rm -f /tmp/workload-images.tar - echo "Saved $(echo "$IMAGES" | wc -l) workload images for cache" + # Use --platform to avoid exporting multi-arch manifest lists + # that reference content not pulled for this platform + docker exec kind-control-plane ctr -n k8s.io images export \ + --platform linux/amd64 /tmp/workload-images.tar $IMAGES || { + echo "Warning: bulk image export failed, trying individual export" + # Fall back to exporting images one by one, skipping failures + > /tmp/image-list-ok.txt + for img in $IMAGES; do + if docker exec kind-control-plane ctr -n k8s.io images export \ + --platform linux/amd64 /tmp/img-single.tar "$img" 2>/dev/null; then + echo "$img" >> /tmp/image-list-ok.txt + else + echo " Skipping $img (export failed)" + fi + docker exec kind-control-plane rm -f /tmp/img-single.tar + done + OK_IMAGES=$(cat /tmp/image-list-ok.txt | tr '\n' ' ') + if [ -n "$OK_IMAGES" ]; then + docker exec kind-control-plane ctr -n k8s.io images export \ + --platform linux/amd64 /tmp/workload-images.tar $OK_IMAGES + fi + } + if docker exec kind-control-plane test -f /tmp/workload-images.tar; then + docker cp kind-control-plane:/tmp/workload-images.tar "$CACHE_DIR/images/workload-images.tar" + docker exec kind-control-plane rm -f /tmp/workload-images.tar + echo "Saved workload images for cache" + else + echo "Warning: no workload images saved for cache" + fi fi fi From 59bd1bfacac097ced601bff3cdd14befd9523b69 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 6 Mar 2026 12:56:17 +0000 Subject: [PATCH 8/9] Fix workload image caching: use host docker pull+save instead of ctr export ctr images export fails on every image in KIND because containerd stores multi-arch manifest lists without all platform layers. Switch to pulling images on the host Docker daemon and using docker save, which produces a tar that kind load image-archive can consume on cache hit. Also make the entire caching block non-fatal (|| true) so image cache failures never break the build. https://claude.ai/code/session_01ReC3SkRaWTNe3BZyuhLRyx Signed-off-by: Claude --- .github/workflows/eval-regression.yaml | 43 ++++++++++---------------- 1 file changed, 16 insertions(+), 27 deletions(-) diff --git a/.github/workflows/eval-regression.yaml b/.github/workflows/eval-regression.yaml index 732a41996f..32c4c0ed57 100644 --- a/.github/workflows/eval-regression.yaml +++ b/.github/workflows/eval-regression.yaml @@ -788,6 +788,8 @@ jobs: wait $METRICS_PID # === Save workload images from KIND for cache (first run only) === + # ctr export doesn't work in KIND (content digest errors on manifest lists), + # so we pull images on the host Docker daemon and use docker save instead. if [ ! -f "$CACHE_DIR/images/workload-images.tar" ]; then echo "Saving workload images for cache (first run only)" IMAGES=$(docker exec kind-control-plane crictl images -o json \ @@ -796,37 +798,24 @@ jobs: | grep -v 'kindest' \ | sort) if [ -n "$IMAGES" ]; then - # Use --platform to avoid exporting multi-arch manifest lists - # that reference content not pulled for this platform - docker exec kind-control-plane ctr -n k8s.io images export \ - --platform linux/amd64 /tmp/workload-images.tar $IMAGES || { - echo "Warning: bulk image export failed, trying individual export" - # Fall back to exporting images one by one, skipping failures - > /tmp/image-list-ok.txt - for img in $IMAGES; do - if docker exec kind-control-plane ctr -n k8s.io images export \ - --platform linux/amd64 /tmp/img-single.tar "$img" 2>/dev/null; then - echo "$img" >> /tmp/image-list-ok.txt - else - echo " Skipping $img (export failed)" - fi - docker exec kind-control-plane rm -f /tmp/img-single.tar - done - OK_IMAGES=$(cat /tmp/image-list-ok.txt | tr '\n' ' ') - if [ -n "$OK_IMAGES" ]; then - docker exec kind-control-plane ctr -n k8s.io images export \ - --platform linux/amd64 /tmp/workload-images.tar $OK_IMAGES + echo "Pulling $(echo "$IMAGES" | wc -l) images on host for caching..." + for img in $IMAGES; do + docker pull "$img" >/dev/null 2>&1 & + done + wait + # Save only images that were successfully pulled + SAVE_LIST="" + for img in $IMAGES; do + if docker image inspect "$img" >/dev/null 2>&1; then + SAVE_LIST="$SAVE_LIST $img" fi - } - if docker exec kind-control-plane test -f /tmp/workload-images.tar; then - docker cp kind-control-plane:/tmp/workload-images.tar "$CACHE_DIR/images/workload-images.tar" - docker exec kind-control-plane rm -f /tmp/workload-images.tar + done + if [ -n "$SAVE_LIST" ]; then + docker save $SAVE_LIST -o "$CACHE_DIR/images/workload-images.tar" echo "Saved workload images for cache" - else - echo "Warning: no workload images saved for cache" fi fi - fi + fi || true # Never fail the build for caching # Wait for background KIND node image save [ -n "${SAVE_NODE_PID:-}" ] && wait $SAVE_NODE_PID || true From 8f7009b21ce15f279b09b8c071334c25c45d32f6 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 7 Mar 2026 20:01:03 +0000 Subject: [PATCH 9/9] Remove orphaned trailing comment in eval-regression workflow https://claude.ai/code/session_01ReC3SkRaWTNe3BZyuhLRyx Signed-off-by: Claude --- .github/workflows/eval-regression.yaml | 1 - 1 file changed, 1 deletion(-) diff --git a/.github/workflows/eval-regression.yaml b/.github/workflows/eval-regression.yaml index 32c4c0ed57..e776aa76c9 100644 --- a/.github/workflows/eval-regression.yaml +++ b/.github/workflows/eval-regression.yaml @@ -1233,4 +1233,3 @@ jobs: else echo "✅ All tests passed." fi -# CI speed optimizations