From b3e477f304e43b7ff6427c7e186bfd9e524bbff2 Mon Sep 17 00:00:00 2001 From: AlexGabbia Date: Tue, 25 Aug 2026 10:24:56 +0200 Subject: [PATCH] fix(update): wait for resumed Windows gateway before failing fleet check MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The post-update fleet version check slept 2s and probed once. On Windows the resume path relaunches the gateway detached, and it needs ~10s to boot (the Telegram polling reconnect) before it stamps gateway_state.json or answers the control socket. That race reported "no rows" for a healthy resume, exited 1, and triggered a full retry that re-killed the gateway the first attempt had just started — leaving it down and surfacing "Update failed (exit 1)". Poll a bounded window (up to 30s) for the resumed gateway to publish its identity, and only treat a persistently empty snapshot as verification failure. The fail-closed contract from #93406 is preserved: a gateway that genuinely never comes back still exits 1. --- contributors/emails/alex.gazzillo@gmail.com | 1 + hermes_cli/update_cmd.py | 43 +++++++++++++++++---- 2 files changed, 37 insertions(+), 7 deletions(-) create mode 100644 contributors/emails/alex.gazzillo@gmail.com diff --git a/contributors/emails/alex.gazzillo@gmail.com b/contributors/emails/alex.gazzillo@gmail.com new file mode 100644 index 0000000000..d87263f046 --- /dev/null +++ b/contributors/emails/alex.gazzillo@gmail.com @@ -0,0 +1 @@ +AlexGabbia diff --git a/hermes_cli/update_cmd.py b/hermes_cli/update_cmd.py index 87c495c817..e44c5bb3bd 100644 --- a/hermes_cli/update_cmd.py +++ b/hermes_cli/update_cmd.py @@ -8685,14 +8685,43 @@ def _cmd_update_impl(args, gateway_mode: bool): # a moment to rewrite gateway_state.json with their new identity. # Skipped when the restart phase touched nothing (no gateways # were running) — nothing to settle. + # + # On Windows the resume path relaunches the gateway DETACHED, and + # that process must boot before it stamps gateway_state.json or + # answers the control socket (a Telegram gateway reconnects its + # polling loop — ~10s). A single 2s sleep therefore races the + # gateway's own startup and reports "no rows" (exit 1) for a + # healthy resume, which then triggers a full retry that re-kills + # the gateway the first attempt just started. Poll a bounded + # window for the resumed gateway to publish its identity instead. + _fleet_snapshot = [] if _fleet_rows_expected: - _time.sleep(2.0) - # Pass the pre-restart PID snapshot so a gateway the restart - # phase stopped WITHOUT a verified replacement shows as a DOWN - # row (exit 1) instead of silently producing no row at all. - _fleet_snapshot = collect_fleet_versions( - pre_restart_pids=_pre_restart_gateway_pids - ) + _fleet_deadline = _time.monotonic() + 30.0 + while True: + _time.sleep(2.0) + # Pass the pre-restart PID snapshot so a gateway the + # restart phase stopped WITHOUT a verified replacement + # shows as a DOWN row (exit 1) instead of silently + # producing no row at all. + _fleet_snapshot = collect_fleet_versions( + pre_restart_pids=_pre_restart_gateway_pids + ) + # A "down" row here is the stale pre-restart record of a + # gateway whose detached replacement is still booting — + # not a confirmed failure. Keep polling until every + # resumed gateway has published (no "down" rows remain) + # or the deadline passes, so a slow second gateway can't + # be misread as down and re-trigger the retry loop. + if _fleet_snapshot and not any( + row.get("state") == "down" for row in _fleet_snapshot + ): + break + if _time.monotonic() >= _fleet_deadline: + break + else: + _fleet_snapshot = collect_fleet_versions( + pre_restart_pids=_pre_restart_gateway_pids + ) if print_fleet_version_matrix(_fleet_snapshot): gateway_fleet_restart_incomplete = True elif not _fleet_snapshot and _fleet_rows_expected: