From 7571ab18adce0a0f533656e7b69e0f93046fbba0 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 04:02:16 -0700 Subject: [PATCH 1/3] fix(gateway): a planned stop stays planned when the watcher beats the CLI's SIGTERM MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `hermes gateway stop` writes the planned-stop marker, then sends SIGTERM. The gateway's planned-stop watcher (0.5 s poll) can fire in between: its shutdown call consumes the marker as planned, and the CLI's SIGTERM that follows finds no marker, is classified as an external kill and the gateway exits 1 — which Restart=on-failure supervisors answer by reviving a gateway the operator just stopped. Remember an accepted planned stop in the handler so the trailing SIGTERM of the same stop is treated as planned. Found by the core parity E2E matrix (gateway/api_server graceful_exit cell), reproduced deterministically by signalling after the watcher has consumed the marker. --- gateway/run.py | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/gateway/run.py b/gateway/run.py index 7d1f33e61703..8ba08d15c52a 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -5264,6 +5264,8 @@ def restart_signal_handler(): def _start_gateway_make_shutdown_signal_handler(runner, _signal_initiated_shutdown: list): """Build the SIGINT/SIGTERM handler; ``_signal_initiated_shutdown[0]`` records an unplanned signal.""" + planned_stop_seen = [False] + def shutdown_signal_handler(received_signal=None): # Planned --replace takeover (sibling marked this PID): exit 0 so systemd won't revive us. def _takeover() -> bool: @@ -5283,6 +5285,12 @@ def _snapshot(): planned_takeover = bool(_best_effort(_takeover, "Takeover marker check failed: %s")) planned_stop = received_signal == signal.SIGINT or ( not planned_takeover and bool(_best_effort(_planned_stop, "Planned stop marker check failed: %s"))) + # `hermes gateway stop` writes the marker, THEN signals: the planned-stop watcher can consume + # the marker in between, and the CLI's own SIGTERM must not then read as an external kill. + if planned_stop: + planned_stop_seen[0] = True + elif planned_stop_seen[0] and not planned_takeover: + planned_stop = True _shutdown_ctx = _best_effort(_snapshot, "snapshot_shutdown_context failed: %s") sig_name = _shutdown_ctx["signal"] if _shutdown_ctx else None From 20864bec552459a5d3adc7c4d380fc6396217e42 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 04:54:16 -0700 Subject: [PATCH 2/3] test(gateway): the CLI SIGTERM after a watcher-consumed planned-stop marker stays planned Invariant for the planned-stop race: the watcher consumes the marker, the CLI trailing SIGTERM must not read as an external kill; a bare SIGTERM with no planned stop before it still does. --- tests/gateway/test_planned_stop_watcher.py | 32 ++++++++++++++++++++++ 1 file changed, 32 insertions(+) diff --git a/tests/gateway/test_planned_stop_watcher.py b/tests/gateway/test_planned_stop_watcher.py index c68ce2c17f8a..f5486cfcaf71 100644 --- a/tests/gateway/test_planned_stop_watcher.py +++ b/tests/gateway/test_planned_stop_watcher.py @@ -201,3 +201,35 @@ def test_planned_stop_marker_targets_self_probe_is_non_destructive(tmp_path, mon assert status_mod.planned_stop_marker_targets_self() is True +def test_cli_sigterm_after_watcher_consumed_the_marker_stays_planned(tmp_path, monkeypatch): + """`hermes gateway stop` writes the marker, THEN signals. When the watcher fires in between and + consumes the marker, the trailing SIGTERM must not read as an external kill (exit 1 → a + Restart=on-failure supervisor revives the gateway the operator stopped). A bare SIGTERM with no + planned stop before it is still an external kill.""" + import signal + from types import SimpleNamespace + + import gateway.run as run_mod + import gateway.shutdown_forensics as forensics + + marker = tmp_path / ".gateway-planned-stop.json" + monkeypatch.setattr(status_mod, "_get_planned_stop_marker_path", lambda: marker) + monkeypatch.setattr(status_mod, "consume_takeover_marker_for_self", lambda: False) + monkeypatch.setattr(forensics, "snapshot_shutdown_context", lambda *a, **k: None) + monkeypatch.setattr(run_mod.asyncio, "create_task", lambda coro: coro.close()) + + def _handler(): + runner = SimpleNamespace(_signal_initiated_shutdown=False, stop=lambda: asyncio.sleep(0)) + flag = [False] + return run_mod._start_gateway_make_shutdown_signal_handler(runner, flag), flag + + handler, flag = _handler() + _write_self_marker(marker) + handler(None) # the watcher's call: consumes the marker + assert not marker.exists() + handler(signal.SIGTERM) # the CLI's own SIGTERM, marker already gone + assert flag[0] is False + + bare, bare_flag = _handler() + bare(signal.SIGTERM) + assert bare_flag[0] is True From 774b68db79ec0e98be39d083fbfdd5150cb1e522 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 05:16:32 -0700 Subject: [PATCH 3/3] chore: retrigger CI (zero-job dispatch failure, auto-heal)