From 609b7bf5108a3bd8c0a66fa5076ce912c9dca90a Mon Sep 17 00:00:00 2001 From: Asher Cohen Date: Sat, 20 Jun 2026 19:06:50 -0700 Subject: [PATCH 1/2] Run ClickHouse CDC setup before analytics consumers start. Production deploys failed when Configure ClickHouse CDC ran while analytics-worker and the metric-stream sink were already loading ClickHouse near its memory cap. Use a temporary Swarm overlay to keep those services at zero replicas through CDC setup, then redeploy the full stack. Co-authored-by: Cursor --- .github/workflows/deploy-web-stack.yml | 94 ++++++++++++++++++++++---- deploy/stack.cdc-quiesce.yml | 12 ++++ 2 files changed, 94 insertions(+), 12 deletions(-) create mode 100644 deploy/stack.cdc-quiesce.yml diff --git a/.github/workflows/deploy-web-stack.yml b/.github/workflows/deploy-web-stack.yml index 8a3e8fc9ec..a6267a2f18 100644 --- a/.github/workflows/deploy-web-stack.yml +++ b/.github/workflows/deploy-web-stack.yml @@ -288,6 +288,8 @@ jobs: run: | node "$RUNNER_TEMP/run-with-dotenv-env.mjs" \ docker stack config $STACK_FILE_FLAGS >/dev/null + node "$RUNNER_TEMP/run-with-dotenv-env.mjs" \ + docker stack config $STACK_FILE_FLAGS -c deploy/stack.cdc-quiesce.yml >/dev/null - name: Login to GHCR uses: docker/login-action@c94ce9fb468520275223c153574b00df6fe4bcc9 # v3 @@ -470,13 +472,13 @@ jobs: cleanup_migration_container echo "Migration succeeded." - - name: Deploy stack - id: deploy_stack + - name: Deploy stack without ClickHouse consumers + id: deploy_stack_quiesced env: IMAGE_TAG: ${{ steps.resolve.outputs.tag }} run: | set +e - timeout 20m node "$RUNNER_TEMP/run-with-dotenv-env.mjs" docker stack deploy $STACK_FILE_FLAGS --with-registry-auth --prune --detach=true "$STACK_NAME" + timeout 20m node "$RUNNER_TEMP/run-with-dotenv-env.mjs" docker stack deploy $STACK_FILE_FLAGS -c deploy/stack.cdc-quiesce.yml --with-registry-auth --prune --detach=true "$STACK_NAME" status=$? set -e if [ "$status" -eq 124 ]; then @@ -487,7 +489,7 @@ jobs: fi wait_deadline_seconds=$((SECONDS + 1200)) - for service in web worker analytics-worker clickhouse; do + for service in web worker analytics-worker metric-stream-clickhouse-sink clickhouse; do service_name="${STACK_NAME}_${service}" while true; do update_state="$( @@ -536,14 +538,6 @@ jobs: done done - - name: Prune obsolete images post-deploy - if: ${{ steps.deploy_stack.outcome == 'success' }} - run: docker image prune -af - - - name: Report Docker disk usage post-prune - if: ${{ steps.deploy_stack.outcome == 'success' }} - run: docker system df - - name: Wait for Postgres writable after stack deploy run: | readiness_attempts=36 @@ -691,6 +685,7 @@ jobs: --type Text - name: Configure ClickHouse CDC + id: configure_clickhouse_cdc env: IMAGE_TAG: ${{ steps.resolve.outputs.tag }} run: | @@ -701,6 +696,81 @@ jobs: "ghcr.io/asherlc/dofek:${IMAGE_TAG}" \ -euc 'export DATABASE_URL="postgres://health:${POSTGRES_PASSWORD}@db:5432/health"; export CLICKHOUSE_URL="http://default:${CLICKHOUSE_PASSWORD_ENCODED}@clickhouse:8123"; exec node --experimental-transform-types --enable-source-maps --disable-warning=ExperimentalWarning src/db/setup-clickhouse-cdc.ts' + - name: Deploy ClickHouse consumer services + id: deploy_stack_full + if: always() && steps.deploy_stack_quiesced.outcome == 'success' + env: + IMAGE_TAG: ${{ steps.resolve.outputs.tag }} + run: | + set +e + timeout 20m node "$RUNNER_TEMP/run-with-dotenv-env.mjs" docker stack deploy $STACK_FILE_FLAGS --with-registry-auth --prune --detach=true "$STACK_NAME" + status=$? + set -e + if [ "$status" -eq 124 ]; then + echo "::error::docker stack deploy exceeded 20m" + fi + if [ "$status" -ne 0 ]; then + exit "$status" + fi + + wait_deadline_seconds=$((SECONDS + 1200)) + for service in analytics-worker metric-stream-clickhouse-sink; do + service_name="${STACK_NAME}_${service}" + while true; do + update_state="$( + docker service inspect "$service_name" \ + --format '{{if .UpdateStatus}}{{.UpdateStatus.State}}{{end}}' + )" + image="$( + docker service inspect "$service_name" \ + --format '{{.Spec.TaskTemplate.ContainerSpec.Image}}' + )" + desired_replicas="$( + docker service inspect "$service_name" \ + --format '{{if .Spec.Mode.Replicated}}{{.Spec.Mode.Replicated.Replicas}}{{end}}' + )" + replicas="$( + docker service ls --filter "name=${service_name}" --format '{{.Replicas}}' + )" + echo "${service_name} image=${image} replicas=${replicas} update_state=${update_state:-none}" + case "$update_state" in + rollback_*) + docker service ps "$service_name" --no-trunc + echo "::error::${service_name} did not finish deployment cleanly; update_state=${update_state}" + exit 1 + ;; + paused) + running="${replicas%%/*}" + if [ "$running" -eq 0 ]; then + docker service ps "$service_name" --no-trunc + echo "::error::${service_name} update is paused with no running tasks" + exit 1 + fi + echo "::warning::${service_name} update is paused but has running tasks; continuing" + ;; + esac + if [ "$replicas" = "${desired_replicas}/${desired_replicas}" ] && { + [ -z "$update_state" ] || [ "$update_state" = "completed" ] + }; then + break + fi + if [ "$SECONDS" -ge "$wait_deadline_seconds" ]; then + docker service ps "$service_name" --no-trunc + echo "::error::${service_name} did not converge before timeout" + exit 124 + fi + sleep 10 + done + done + + - name: Prune obsolete images post-deploy + if: ${{ steps.deploy_stack_full.outcome == 'success' }} + run: docker image prune -af + + - name: Report Docker disk usage post-prune + if: ${{ steps.deploy_stack_full.outcome == 'success' }} + run: docker system df + - uses: ./.github/actions/append-dispatcher-run-summary if: always() with: diff --git a/deploy/stack.cdc-quiesce.yml b/deploy/stack.cdc-quiesce.yml new file mode 100644 index 0000000000..bd34e60cb7 --- /dev/null +++ b/deploy/stack.cdc-quiesce.yml @@ -0,0 +1,12 @@ +# Temporary deploy overlay: keep heavy ClickHouse consumers at zero replicas +# while post-deploy CDC setup runs. The deploy workflow applies this overlay +# for the first stack deploy, runs Configure ClickHouse CDC, then redeploys +# without this file so analytics-worker and the metric-stream sink start. +services: + analytics-worker: + deploy: + replicas: 0 + + metric-stream-clickhouse-sink: + deploy: + replicas: 0 From 5e9661832d114c37d60550b76d96f28945d077cc Mon Sep 17 00:00:00 2001 From: Asher Cohen Date: Sat, 20 Jun 2026 19:22:57 -0700 Subject: [PATCH 2/2] Address deploy workflow review feedback. Run the consumer restore deploy whenever the quiesced deploy step ran, even if it failed, so analytics-worker is not left at zero replicas. Fail the final convergence loop on paused Swarm updates instead of warning and continuing. Co-authored-by: Cursor --- .github/workflows/deploy-web-stack.yml | 12 ++++-------- 1 file changed, 4 insertions(+), 8 deletions(-) diff --git a/.github/workflows/deploy-web-stack.yml b/.github/workflows/deploy-web-stack.yml index a6267a2f18..4bb0fc440a 100644 --- a/.github/workflows/deploy-web-stack.yml +++ b/.github/workflows/deploy-web-stack.yml @@ -698,7 +698,7 @@ jobs: - name: Deploy ClickHouse consumer services id: deploy_stack_full - if: always() && steps.deploy_stack_quiesced.outcome == 'success' + if: always() && steps.deploy_stack_quiesced.conclusion != 'skipped' env: IMAGE_TAG: ${{ steps.resolve.outputs.tag }} run: | @@ -740,13 +740,9 @@ jobs: exit 1 ;; paused) - running="${replicas%%/*}" - if [ "$running" -eq 0 ]; then - docker service ps "$service_name" --no-trunc - echo "::error::${service_name} update is paused with no running tasks" - exit 1 - fi - echo "::warning::${service_name} update is paused but has running tasks; continuing" + docker service ps "$service_name" --no-trunc + echo "::error::${service_name} update is paused" + exit 1 ;; esac if [ "$replicas" = "${desired_replicas}/${desired_replicas}" ] && {