From e075e057e70619fa45eecb9faec2515dfe7524d4 Mon Sep 17 00:00:00 2001 From: Ari Heber Date: Sun, 30 Aug 2026 14:24:59 +0300 Subject: [PATCH 1/4] Roll passthrough-watcher pod when its ConfigMap changes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit watch.sh reads mode from a file each loop, but HEALTH_URL is a plain shell variable set once at process start from the ConfigMap-baked script text. A ConfigMap update (e.g. the healthPath fix in #56) never reaches an already-running watcher pod without a restart, so any tenant whose pod predated that fix was silently stuck polling /health/live — which always returns 200 — making both the manual bypass toggle and automatic dependency-outage failover no-ops. Add a checksum/config annotation on the pod template so config changes trigger a real rollout. --- charts/yuki/templates/passthrough-watcher-deployment.yaml | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/charts/yuki/templates/passthrough-watcher-deployment.yaml b/charts/yuki/templates/passthrough-watcher-deployment.yaml index d99ed7d..3d04d3e 100644 --- a/charts/yuki/templates/passthrough-watcher-deployment.yaml +++ b/charts/yuki/templates/passthrough-watcher-deployment.yaml @@ -20,6 +20,10 @@ spec: labels: app: {{ .Values.app.name }}-passthrough-watcher group: {{ .Values.app.group }} + annotations: + # Force a rollout whenever mode/healthPath/etc. change — watch.sh reads HEALTH_URL + # once at process start, so a ConfigMap update alone never reaches a running pod. + checksum/config: {{ include (print $.Template.BasePath "/passthrough-watcher-configmap.yaml") . | sha256sum }} spec: serviceAccountName: {{ .Values.app.name }}-passthrough-watcher securityContext: From 257b53d60276711176fcd900320976f48754a685 Mon Sep 17 00:00:00 2001 From: Ari Heber Date: Sun, 30 Aug 2026 14:29:31 +0300 Subject: [PATCH 2/4] Live-reload healthPath instead of forcing a pod restart MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit HEALTH_URL was baked in once at process start, so a ConfigMap change (like #56's /health/live -> /health/passthrough fix) never reached an already-running watcher pod. Re-read healthPath from a file each loop, the same way mode already is, so a values change (or this ConfigMap already having drifted ahead of a stale pod) takes effect within one poll interval — no restart, no checksum annotation needed. --- charts/yuki/templates/passthrough-watcher-configmap.yaml | 7 +++++-- charts/yuki/templates/passthrough-watcher-deployment.yaml | 4 ---- 2 files changed, 5 insertions(+), 6 deletions(-) diff --git a/charts/yuki/templates/passthrough-watcher-configmap.yaml b/charts/yuki/templates/passthrough-watcher-configmap.yaml index f74b4d8..036317d 100644 --- a/charts/yuki/templates/passthrough-watcher-configmap.yaml +++ b/charts/yuki/templates/passthrough-watcher-configmap.yaml @@ -9,6 +9,7 @@ metadata: data: # Re-read each loop iteration, so a values change applies live. mode: {{ .Values.passthrough.mode | quote }} + healthPath: {{ .Values.passthrough.watcher.healthPath | quote }} watch.sh: | #!/bin/sh set -eu -o pipefail @@ -19,7 +20,7 @@ data: API=https://kubernetes.default.svc SVC_URL="$API/api/v1/namespaces/$NAMESPACE/services/{{ .Values.app.name }}" # Not the Service above — that's the one we swap, checking through it would be circular. - HEALTH_URL="http://{{ .Values.app.name }}-direct:{{ .Values.app.service.port }}{{ .Values.passthrough.watcher.healthPath }}" + HEALTH_BASE="http://{{ .Values.app.name }}-direct:{{ .Values.app.service.port }}" PROXY_LABEL="{{ .Values.app.name }}" PASSTHROUGH_LABEL="{{ .Values.app.name }}-passthrough" FAILURE_THRESHOLD={{ .Values.passthrough.watcher.failureThreshold }} @@ -66,10 +67,12 @@ data: fail_count=0 success_count=0 - echo "$(date -Iseconds) bypass-watcher: started, mode=$(cat /etc/bypass/mode 2>/dev/null || echo auto) health_url=$HEALTH_URL" + echo "$(date -Iseconds) bypass-watcher: started, mode=$(cat /etc/bypass/mode 2>/dev/null || echo auto) health_base=$HEALTH_BASE" while true; do mode=$(cat /etc/bypass/mode 2>/dev/null || echo auto) + health_path=$(cat /etc/bypass/healthPath 2>/dev/null || echo /health/passthrough) + HEALTH_URL="$HEALTH_BASE$health_path" # Re-read the real Service each loop instead of trusting an in-memory value — # a stale/wrong belief here is worse than one extra API call per poll. diff --git a/charts/yuki/templates/passthrough-watcher-deployment.yaml b/charts/yuki/templates/passthrough-watcher-deployment.yaml index 3d04d3e..d99ed7d 100644 --- a/charts/yuki/templates/passthrough-watcher-deployment.yaml +++ b/charts/yuki/templates/passthrough-watcher-deployment.yaml @@ -20,10 +20,6 @@ spec: labels: app: {{ .Values.app.name }}-passthrough-watcher group: {{ .Values.app.group }} - annotations: - # Force a rollout whenever mode/healthPath/etc. change — watch.sh reads HEALTH_URL - # once at process start, so a ConfigMap update alone never reaches a running pod. - checksum/config: {{ include (print $.Template.BasePath "/passthrough-watcher-configmap.yaml") . | sha256sum }} spec: serviceAccountName: {{ .Values.app.name }}-passthrough-watcher securityContext: From da1b52f4b885656d9b6eadcc49e5f8153bb04fb7 Mon Sep 17 00:00:00 2001 From: Ari Heber Date: Sun, 30 Aug 2026 14:31:08 +0300 Subject: [PATCH 3/4] Revert "Live-reload healthPath instead of forcing a pod restart" This reverts commit 257b53d60276711176fcd900320976f48754a685. --- charts/yuki/templates/passthrough-watcher-configmap.yaml | 7 ++----- charts/yuki/templates/passthrough-watcher-deployment.yaml | 4 ++++ 2 files changed, 6 insertions(+), 5 deletions(-) diff --git a/charts/yuki/templates/passthrough-watcher-configmap.yaml b/charts/yuki/templates/passthrough-watcher-configmap.yaml index 036317d..f74b4d8 100644 --- a/charts/yuki/templates/passthrough-watcher-configmap.yaml +++ b/charts/yuki/templates/passthrough-watcher-configmap.yaml @@ -9,7 +9,6 @@ metadata: data: # Re-read each loop iteration, so a values change applies live. mode: {{ .Values.passthrough.mode | quote }} - healthPath: {{ .Values.passthrough.watcher.healthPath | quote }} watch.sh: | #!/bin/sh set -eu -o pipefail @@ -20,7 +19,7 @@ data: API=https://kubernetes.default.svc SVC_URL="$API/api/v1/namespaces/$NAMESPACE/services/{{ .Values.app.name }}" # Not the Service above — that's the one we swap, checking through it would be circular. - HEALTH_BASE="http://{{ .Values.app.name }}-direct:{{ .Values.app.service.port }}" + HEALTH_URL="http://{{ .Values.app.name }}-direct:{{ .Values.app.service.port }}{{ .Values.passthrough.watcher.healthPath }}" PROXY_LABEL="{{ .Values.app.name }}" PASSTHROUGH_LABEL="{{ .Values.app.name }}-passthrough" FAILURE_THRESHOLD={{ .Values.passthrough.watcher.failureThreshold }} @@ -67,12 +66,10 @@ data: fail_count=0 success_count=0 - echo "$(date -Iseconds) bypass-watcher: started, mode=$(cat /etc/bypass/mode 2>/dev/null || echo auto) health_base=$HEALTH_BASE" + echo "$(date -Iseconds) bypass-watcher: started, mode=$(cat /etc/bypass/mode 2>/dev/null || echo auto) health_url=$HEALTH_URL" while true; do mode=$(cat /etc/bypass/mode 2>/dev/null || echo auto) - health_path=$(cat /etc/bypass/healthPath 2>/dev/null || echo /health/passthrough) - HEALTH_URL="$HEALTH_BASE$health_path" # Re-read the real Service each loop instead of trusting an in-memory value — # a stale/wrong belief here is worse than one extra API call per poll. diff --git a/charts/yuki/templates/passthrough-watcher-deployment.yaml b/charts/yuki/templates/passthrough-watcher-deployment.yaml index d99ed7d..3d04d3e 100644 --- a/charts/yuki/templates/passthrough-watcher-deployment.yaml +++ b/charts/yuki/templates/passthrough-watcher-deployment.yaml @@ -20,6 +20,10 @@ spec: labels: app: {{ .Values.app.name }}-passthrough-watcher group: {{ .Values.app.group }} + annotations: + # Force a rollout whenever mode/healthPath/etc. change — watch.sh reads HEALTH_URL + # once at process start, so a ConfigMap update alone never reaches a running pod. + checksum/config: {{ include (print $.Template.BasePath "/passthrough-watcher-configmap.yaml") . | sha256sum }} spec: serviceAccountName: {{ .Values.app.name }}-passthrough-watcher securityContext: From fb4112223caed45fe5c83e0a0b9246544a94a704 Mon Sep 17 00:00:00 2001 From: Ari Heber Date: Sun, 30 Aug 2026 14:47:25 +0300 Subject: [PATCH 4/4] Trim checksum annotation comment to one line --- charts/yuki/templates/passthrough-watcher-deployment.yaml | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/charts/yuki/templates/passthrough-watcher-deployment.yaml b/charts/yuki/templates/passthrough-watcher-deployment.yaml index 3d04d3e..2424942 100644 --- a/charts/yuki/templates/passthrough-watcher-deployment.yaml +++ b/charts/yuki/templates/passthrough-watcher-deployment.yaml @@ -21,8 +21,7 @@ spec: app: {{ .Values.app.name }}-passthrough-watcher group: {{ .Values.app.group }} annotations: - # Force a rollout whenever mode/healthPath/etc. change — watch.sh reads HEALTH_URL - # once at process start, so a ConfigMap update alone never reaches a running pod. + # Rolls the pod whenever the ConfigMap changes — watch.sh only reads it at start. checksum/config: {{ include (print $.Template.BasePath "/passthrough-watcher-configmap.yaml") . | sha256sum }} spec: serviceAccountName: {{ .Values.app.name }}-passthrough-watcher