From cf01bd3f15a5ce4471f4479f8e2c5b18cb6b79a9 Mon Sep 17 00:00:00 2001 From: omer-vishlitzky Date: Thu, 11 Jun 2026 07:06:38 +0300 Subject: [PATCH] NO-ISSUE: deploy fulfillment pods at zero replicas during kustomize apply MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The kustomize overlay changes the fulfillment-database StatefulSet image reference from a :latest tag to a @sha256: digest. This triggers a StatefulSet pod recreation, killing the database mid-connection. If the grpc-server is running database migrations at that moment, golang-migrate leaves the schema_migrations table in a dirty state and all subsequent grpc-server starts refuse to run — causing a boot failure. Fix: use `kustomize edit set replicas` to set fulfillment-controller and fulfillment-grpc-server to 0 before applying the overlay. The apply itself deploys with zero replicas, eliminating the race entirely. At step [5/9], the script now: 1. Waits for TLS certificates 2. Waits for the database StatefulSet rollout to complete 3. Scales grpc-server and controller back to 1 4. Waits for all fulfillment deployment rollouts This ensures the database is healthy before any migration-running pod starts, and that grpc-server is available before rest-gateway's readiness probe checks the gRPC upstream. Depends on: openshift/release#80431 (adds kustomize binary to the installer container image). Co-Authored-By: Claude Opus 4.6 (1M context) --- scripts/refresh-after-snapshot.sh | 26 +++++++++++++++++++------- 1 file changed, 19 insertions(+), 7 deletions(-) diff --git a/scripts/refresh-after-snapshot.sh b/scripts/refresh-after-snapshot.sh index 8f095c52..5a243b08 100755 --- a/scripts/refresh-after-snapshot.sh +++ b/scripts/refresh-after-snapshot.sh @@ -164,7 +164,6 @@ failed=0 wait ${pid_creds} || failed=1 if (( failed )); then echo "ERROR: Failed to create fulfillment credentials"; exit 1; fi -oc scale deploy/fulfillment-controller -n "${INSTALLER_NAMESPACE}" --replicas=0 2>/dev/null || true echo "[3/9] Applying kustomize overlay..." oc delete job -n "${INSTALLER_NAMESPACE}" --all --ignore-not-found # Exclude only the bootstrap job — it's redundant on snapshot boot and races @@ -173,8 +172,14 @@ oc delete job -n "${INSTALLER_NAMESPACE}" --all --ignore-not-found # operator triggers into one reconciliation instead of a separate oc patch. sed '/job\.yaml/d' base/osac-aap/config/base/kustomization.yaml > base/osac-aap/config/base/kustomization.yaml.tmp \ && mv base/osac-aap/config/base/kustomization.yaml.tmp base/osac-aap/config/base/kustomization.yaml +# Deploy with fulfillment pods scaled to zero. The apply changes the database +# StatefulSet image ref (tag → digest), triggering a pod recreation. If the +# grpc-server were running, it could be mid-migration when the database is +# killed, leaving golang-migrate's schema dirty. Deploying at zero replicas +# eliminates the race entirely — pods are brought back at step [5/9] after +# the database rollout completes. +( cd "overlays/${INSTALLER_KUSTOMIZE_OVERLAY}" && kustomize edit set replicas fulfillment-controller=0 fulfillment-grpc-server=0 ) oc apply -k "overlays/${INSTALLER_KUSTOMIZE_OVERLAY}" -oc scale deploy/fulfillment-controller -n "${INSTALLER_NAMESPACE}" --replicas=0 2>/dev/null || true REPO_ROOT="$(cd "${SCRIPT_DIR}/.." && pwd -P)" PULL_SECRET="${REPO_ROOT}/overlays/${INSTALLER_KUSTOMIZE_OVERLAY}/files/quay-pull-secret.json" @@ -306,12 +311,20 @@ for cert in "${fs_certs[@]}"; do -n "${INSTALLER_NAMESPACE}" --timeout=300s & pids+=($!) done +failed=0 +for pid in "${pids[@]}"; do wait "${pid}" || failed=1; done +if (( failed )); then echo "ERROR: TLS certificates not ready"; exit 1; fi +# Wait for the database StatefulSet rollout before scaling up pods that run +# migrations. The kustomize apply changed the image ref, triggering a +# recreation — the database must be accepting connections before grpc-server +# starts. +oc rollout status statefulset/fulfillment-database -n "${INSTALLER_NAMESPACE}" --timeout=300s +oc scale deploy/fulfillment-controller -n "${INSTALLER_NAMESPACE}" --replicas=1 +oc scale deploy/fulfillment-grpc-server -n "${INSTALLER_NAMESPACE}" --replicas=1 # Envoy never re-reads its config after startup. Restart ingress-proxy before # the rollout wait so that pods depending on new routes (e.g. JWKS) can start. oc rollout restart deploy/fulfillment-ingress-proxy -n "${INSTALLER_NAMESPACE}" -# Kustomize apply may have changed deployment images, triggering new rollouts -# that run DB migrations. Wait for those to finish before restarting pods — -# otherwise the restart kills pods mid-migration and leaves the DB dirty. +pids=() for deploy in "${FULFILLMENT_DEPLOYS[@]}"; do oc rollout status "deploy/${deploy}" -n "${INSTALLER_NAMESPACE}" --timeout=300s & pids+=($!) @@ -319,9 +332,8 @@ done failed=0 for pid in "${pids[@]}"; do wait "${pid}" || failed=1; done wait ${pid_cdi} || failed=1 -if (( failed )); then echo "ERROR: TLS certificates or fulfillment rollouts not ready"; exit 1; fi +if (( failed )); then echo "ERROR: Fulfillment rollouts not ready"; exit 1; fi echo "[5/9] TLS certificates ready, restarting pods..." -oc scale deploy/fulfillment-controller -n "${INSTALLER_NAMESPACE}" --replicas=1 for deploy in "${FULFILLMENT_DEPLOYS[@]}"; do oc rollout restart "deploy/${deploy}" -n "${INSTALLER_NAMESPACE}" done