"""Gateway fleet restart + post-update verification for ``hermes update``. Split out of ``hermes_cli/update_cmd.py``; every name is re-imported there so ``hermes_cli.update_cmd.`` keeps resolving/monkeypatching. Origin helpers are imported lazily inside each function (no import cycle; test patches stay effective). """ import json import logging import re from contextlib import suppress import os import subprocess import sys import time as _time from dataclasses import dataclass, field from pathlib import Path from hermes_cli.update_cmd_common import _best_effort from hermes_cli.update_inventory import _gateway_service_matches_profile # Log-record parity with the origin module. logger = logging.getLogger("hermes_cli.update_cmd") # Under HERMES_HOME (not next to the venv): records the fleet-restart obligation # after a pull advanced HEAD; cleared only when the restart completes or nothing ran. # The existing ``.update-incomplete`` / ``.lazy-refresh-incomplete`` markers gate dependency/venv repair; # this one is the fleet-restart obligation after a git pull that advanced HEAD (#95294). _FLEET_RESTART_PENDING_NAME = "fleet_restart_pending" _FRESH_RESTART_SUPERVISORS = frozenset({"systemd", "launchd", "service", "s6"}) # A supervisor can report a restarted unit active before the gateway finishes its # bootstrap and publishes ``gateway_state.json``. Keep the readiness poll bounded, # but allow the default systemd startup budget plus status-publication slack. _FLEET_PROBE_SETTLE_TIMEOUT_SECONDS = 120.0 _SYSTEMD_SCOPES = (("user", ["systemctl", "--user"]), ("system", ["systemctl"])) _LIST_GATEWAY_UNITS = [ "list-units", "hermes-gateway*", "hermes-serve*", "hermes-dashboard*", "--plain", "--no-legend", "--no-pager", ] def _write_gateway_update_exit_code(ok: bool) -> None: from hermes_cli.update_cmd import get_hermes_home path = get_hermes_home() / ".update_exit_code" with suppress(OSError): path.write_text("0" if ok else "1", encoding="utf-8") def _fleet_restart_pending_marker_path() -> Path: """LEGACY per-``HERMES_HOME`` breadcrumb. Read-compat only — nothing writes it any more. One host runs one multiplexing gateway, so the pull→restart obligation is host-scoped (``hermes_cli/update_host_obligation.py``). An obligation armed by the old per-profile code is still read and cleared here so an in-flight update is discharged after the upgrade. """ from hermes_cli.update_cmd import get_hermes_home return get_hermes_home() / _FLEET_RESTART_PENDING_NAME def _write_legacy_fleet_restart_pending_marker( *, expected_sha: str = "", runtimes: list[dict] | None = None ) -> bool: """Arm the LEGACY per-``HERMES_HOME`` marker. True when written. Never raises. Fallback only: ``$HERMES_HOME`` is writable by construction (the updater already writes its receipts there), so it still carries the obligation when the host state dir cannot. """ path = _fleet_restart_pending_marker_path() try: lines = [f"started={_time.time()}", f"pid={os.getpid()}"] if expected_sha: lines.append(f"expected_sha={expected_sha}") if runtimes is not None: lines.append("inventory=" + json.dumps({"version": 1, "runtimes": runtimes})) path.write_text("\n".join(lines) + "\n", encoding="utf-8") return True except OSError as exc: logger.debug("Could not write legacy fleet-restart-pending marker: %s", exc) return False def _write_fleet_restart_pending_marker(*, expected_sha: str = "", runtimes: list[dict] | None = None) -> None: """Arm the HOST pull→restart obligation. Never raises. An unwritable host state dir (``HERMES_GATEWAY_LOCK_DIR`` on a read-only mount, a container UID that does not own ``$HOME``) must never disarm the obligation: an update interrupted after this point would then leave stale code running with no warning and no catch-up restart (#117275). The legacy per-home marker — which every reader here still honours — carries it instead, and a host that can write neither says so out loud. """ if runtimes == []: # An explicit empty inventory owes no restart (e.g. Desktop-hosted `serve` with no # gateway services). Arming the marker here leaves a breadcrumb nothing can discharge: # a no-gateway host would then fail every later ``hermes update`` (#115311). return from hermes_cli.update_cmd import _m from hermes_cli.update_host_obligation import host_obligation_path, write_host_obligation if _m()._pytest_owns_live_checkout(_fleet_restart_pending_marker_path().parent): logger.debug("Skipping fleet-restart-pending obligation under pytest (live checkout)") return if write_host_obligation( expected_sha=expected_sha, runtimes=runtimes, profile=_current_profile_name()): return if _write_legacy_fleet_restart_pending_marker(expected_sha=expected_sha, runtimes=runtimes): logger.warning( "Host update-restart obligation (%s) is unwritable; armed the per-home marker %s instead.", host_obligation_path(), _fleet_restart_pending_marker_path()) return logger.error( "Could not arm the update-restart obligation in %s or %s; an interrupted update will not warn.", host_obligation_path(), _fleet_restart_pending_marker_path()) print( " ⚠ Could not record the pending gateway-restart obligation (state dir not writable) — " "restart gateways with `hermes gateway restart` if this update is interrupted.", file=sys.stderr, ) def _current_profile_name() -> str: """Profile whose CLI armed the obligation (diagnostics only — the record is host-scoped).""" try: from hermes_cli.profiles import get_active_profile_name return get_active_profile_name() or "default" except Exception: return "" def _clear_fleet_restart_pending_marker() -> None: """Discharge the obligation for the whole host (legacy per-home marker included). Never raises.""" from hermes_cli.update_cmd import _m from hermes_cli.update_host_obligation import clear_host_obligation clear_host_obligation() _m()._clear_marker_file(_fleet_restart_pending_marker_path(), label="fleet-restart-pending") def _fleet_restart_obligation_armed() -> bool: """True when this HOST owes a fleet restart — from any profile's CLI.""" from hermes_cli.update_host_obligation import host_obligation_present if host_obligation_present(): return True with suppress(OSError): return _fleet_restart_pending_marker_path().is_file() return False def _obligation_fields() -> dict[str, str] | None: """Armed obligation as ``key=value`` fields: HOST record first, then the legacy marker. ``None`` means nothing armed OR a malformed record; both must leave the obligation standing. """ from hermes_cli.update_host_obligation import host_obligation_present, obligation_fields fields = obligation_fields() if fields is not None: return fields if host_obligation_present(): # The record exists but its terms are unknown (corrupt, or a NEWER CLI's version). An # unrelated legacy marker's inventory cannot discharge terms nobody can read: fail closed. return None try: text = _fleet_restart_pending_marker_path().read_text(encoding="utf-8-sig") except (OSError, UnicodeError): return None legacy: dict[str, str] = {} for line in text.splitlines(): key, sep, value = line.partition("=") if not sep or key in legacy: return None legacy[key] = value return legacy def _current_checkout_sha() -> str | None: """Current on-disk checkout HEAD, or None if it cannot be resolved.""" from hermes_cli.update_cmd import _capture_head_sha, _m try: from hermes_cli.version_info import get_code_identity sha = (get_code_identity(refresh=True) or {}).get("sha") return str(sha) if sha else None except Exception: return _capture_head_sha(["git"], _m().PROJECT_ROOT) def _receipt_looks_unfinished(receipt: dict) -> bool: """True when *receipt* is from an update that did not finish cleanly. The command boundary stamps a ``stop_reason`` on every receipt, including clean ones (``completed at command boundary``, ``sys.exit(0)``); it must not make a successful receipt look unfinished, or the next ``hermes update`` retriggers ``fleet_restart_pending`` from pre-pull plan SHAs (#98022). """ exit_code = receipt.get("exit_code") outcome = receipt.get("outcome") if exit_code not in (0, None) or outcome in ("failed", "partial", "running"): return True gateway_restart = receipt.get("gateway_restart") if isinstance(gateway_restart, dict) and gateway_restart.get("incomplete"): return True # A stop_reason alone (update_contract refusals: outcome="refused", no exit_code) # counts only when nothing else vouched for success. succeeded = exit_code == 0 or outcome == "success" return bool(receipt.get("stop_reason")) and not succeeded def _receipt_reports_stale_runtime(receipt: dict, expected_sha: str | None = None) -> bool: """True when ``update_receipts/latest.json`` records a runtime SHA skew. Prefer the post-restart ``fleet`` matrix. ``plan.runtimes[].code_sha`` is captured *before* the pull, so a finished update's plan always looks stale and must not retrigger a restart; consult it only for an unfinished receipt. See #95294. """ from hermes_cli.update_cmd import _current_checkout_sha if not isinstance(receipt, dict): return False expected_sha = expected_sha or _current_checkout_sha() if not expected_sha: return False def _sha_mismatch(code_sha) -> bool: return bool(code_sha) and str(code_sha) != str(expected_sha) from hermes_cli.update_receipt import row_is_external fleet = receipt.get("fleet") if isinstance(fleet, list) and fleet: return any( isinstance(entry, dict) and not row_is_external(entry) and (entry.get("state") == "stale" or _sha_mismatch(entry.get("code_sha"))) for entry in fleet ) if not _receipt_looks_unfinished(receipt): return False plan = receipt.get("plan") if not isinstance(plan, dict): return False return any( isinstance(runtime, dict) and _sha_mismatch(runtime.get("code_sha")) for runtime in plan.get("runtimes") or [] ) _SUPERVISED_SERVE_BACKENDS = frozenset( {"manual-serve", "desktop", "desktop-ssh", "systemd", "launchd", "windows-service", "service"} ) # Backends whose supervisor restarts the process without any updater bookkeeping. ``manual-serve`` # is excluded: it owes a durable handoff (``defer_manual_serve``) before it stops counting. # ``systemd``/``windows-service``/``service`` mirror ``_SUPERVISED_SERVE_BACKENDS`` for parity only — # the inventory writer classifies a serve/dashboard row as exactly launchd, desktop, desktop-ssh or manual-serve # (``update_inventory._collect_ledger_runtimes``); those three are set for gateway rows alone. _SUPERVISOR_OWNED_SERVE_BACKENDS = _SUPERVISED_SERVE_BACKENDS - {"manual-serve"} def _receipt_owed_gateways(receipt: dict, pending_manual: list[dict]) -> set[tuple[str, str]] | None: """Pure coverage classification after manual retention of this receipt snapshot. Empty means this receipt owes no gateways, not that an independent marker owes none. Unknown identities, unclassified serve backends and failed manual transfers make coverage unverified. """ plan = receipt.get("plan") or {} entries: list[tuple[object, str | None]] = [(entry, None) for entry in plan.get("runtimes") or []] entries.extend((entry, None) for entry in receipt.get("pending_manual_serves") or []) entries.extend((entry, "gateway") for entry in receipt.get("fleet") or []) owed: set[tuple[str, str]] = set() unverified = False for entry, default_kind in entries: if not isinstance(entry, dict): unverified = True continue kind = entry.get("kind", default_kind) profile = entry.get("profile") # A serve/dashboard row is outside the gateway matrix's evidence, not evidence against the # gateways it does cover: a supervised backend (desktop, systemd, launchd) is its # supervisor's to restart, and a manual-serve row outside the retention list is the serve # obligation mechanism's — a host running a dashboard carries such a row in every receipt, # and a blanket veto made the gateway warning permanently undischargeable there (#115090). # Only an unclassified backend or a failed manual transfer still makes coverage unverified. if kind in ("serve", "dashboard") and entry.get("supervisor") in _SUPERVISED_SERVE_BACKENDS and entry not in pending_manual: continue if kind != "gateway" or not profile or profile == "unknown": unverified = True continue owed.add((kind, profile)) return None if unverified else owed def _fleet_covered_gateways(fleet: list) -> set[tuple[str, str]] | None: """``(kind, profile)`` identities the live rows vouch for; ``None`` when any row is unidentified. A multiplexer's row carries the ``served_profiles`` its runtime status records (``_fleet_row`` keeps the field only when well-formed), so one live process covers every profile it serves. """ covered: set[tuple[str, str]] = set() for row in fleet: profile = row.get("profile") if isinstance(row, dict) else None if not profile or profile == "unknown": return None # unidentified runtime: the matrix cannot vouch for it covered.add(("gateway", profile)) covered.update(("gateway", served) for served in row.get("served_profiles") or []) return covered def _live_fleet_covers_receipt(expected_sha: str | None, receipt: dict, owed: set[tuple[str, str]] | None, *, accept_states: tuple = ("current",)) -> bool: """Require a successor at the expected SHA for every owed gateway identity.""" if not expected_sha: return False from hermes_cli.update_receipt import collect_fleet_versions, row_is_external try: if owed is None: return False if not owed: return bool((receipt.get("plan") or {}).get("runtimes")) fleet = collect_fleet_versions() # State labels are checkout-relative; completed restarts may accept stale rows at the pulled SHA. if not fleet or any( row.get("state") not in accept_states or row.get("code_sha") != expected_sha for row in fleet if not row_is_external(row) ): return False covered = _fleet_covered_gateways(fleet) return covered is not None and owed <= covered except Exception as exc: logger.debug("Could not reconcile pending fleet identities: %s", exc) return False def _marker_owed_gateways(inventory: object) -> set[tuple[str, str]] | None: """The ``("gateway", profile)`` set a marker's inventory owes; None when it recorded none. Raises ValueError for a malformed or unsupported inventory, which keeps the marker. """ from hermes_cli.update_cmd_fleet_gatewayless import runtime_outside_gateway_evidence if inventory is None: return None if not isinstance(inventory, dict) or inventory.get("version") != 1: raise ValueError("unsupported fleet-restart inventory") runtimes = inventory.get("runtimes") if not isinstance(runtimes, list): raise ValueError("fleet-restart inventory has no runtime list") owed: set[tuple[str, str]] = set() for runtime in runtimes: if not isinstance(runtime, dict): raise ValueError("fleet-restart inventory row is not an object") if runtime_outside_gateway_evidence(runtime): continue profile = runtime.get("profile") if runtime.get("kind") != "gateway" or not isinstance(profile, str) or not profile.strip() or profile == "unknown": raise ValueError("fleet-restart inventory row is not an identified gateway") owed.add(("gateway", profile)) return owed def _discharge_gatewayless_marker(checkout_sha: str, expected_sha: str) -> bool: """Settle an inventory-less marker on a host with no live gateway (#118742). Only when the host itself shows nothing the update could still owe a restart to, and HEAD still holds the code it pulled. """ from hermes_cli.update_cmd_fleet_checkout import checkout_contains from hermes_cli.update_cmd_fleet_gatewayless import host_owes_no_gateway_restart try: gatewayless = (checkout_sha == expected_sha or checkout_contains(expected_sha)) and host_owes_no_gateway_restart() except Exception as exc: logger.debug("Gateway-less host probe failed; keeping fleet-restart-pending marker: %s", exc) return False if not gatewayless: return False _clear_fleet_restart_pending_marker() logger.debug("Fleet-restart-pending marker discharged: host runs no gateway at %s", checkout_sha[:10]) return True def _marker_only_restart_obsolete() -> bool: """Settle only the inventory stored with this marker's target SHA. Historical receipts cannot narrow this obligation. Malformed or unsupported inventories stay fail-closed; empty discovery never proves a stopped gateway recovered. Two shapes record no obligation and settle without one: an explicit empty inventory (a pull that found no gateway, #115311) clears outright, and an inventory-less marker (the pre-inventory writer, or a tail that died before its inventory was recorded, #115638) clears once every live gateway is current on the checkout — there is no recorded owed set, so the fleet running the code on disk is the whole of the evidence the marker's warning can be about, even after HEAD moved past ``expected_sha`` by an out-of-band pull — and so does an inventory-less record armed with no SHA at all (a no-op update whose head capture failed, #125952): with no owed set and no SHA, the checkout is the only code it can be held to. With no live gateway at all, the inventory-less marker asks the host instead (``update_cmd_fleet_gatewayless``): it clears when no profile left a gateway that should be running and every live runtime is supervisor-owned or handed off, so a Desktop-only install stops failing every later update (#118742). A serve/dashboard row whose supervisor owns the restart (Desktop backend, systemd/launchd unit, Windows service) is outside the gateway matrix's evidence, not evidence against it — the same boundary ``_receipt_owed_gateways`` draws for receipts (#115090) and the restart phase draws for the Desktop backend (#111494). Counting it made the warning permanently undischargeable on every host that runs a dashboard. A manual-serve row still needs its durable handoff (``defer_manual_serve``), and an unclassified backend stays fail-closed. Discharging here strands nobody: the same row is still accounted at update time by ``update_inventory.report_unaccounted_runtimes``, which prints it and exits 1 when the restart phase never touched it — this marker only stops re-warning about it on every later startup. """ from hermes_cli.update_cmd_fleet_checkout import checkout_contains try: fields = _obligation_fields() if fields is None: return False expected_sha = fields.get("expected_sha", "").strip() owed = _marker_owed_gateways(json.loads(fields.get("inventory", "null"))) except (OSError, UnicodeError, ValueError): return False if owed is not None and not owed: # A pull that recorded no gateway runtime owes no restart; clearing avoids the # stuck "Fleet restart incomplete" loop on Desktop-hosted (no-service) installs. _clear_fleet_restart_pending_marker() logger.debug("Fleet-restart-pending marker discharged: no gateway obligation recorded") return True if owed is not None and not expected_sha: return False # an inventoried obligation without its SHA can never be proven checkout_sha = _current_checkout_sha() if owed is not None and checkout_sha != expected_sha and not checkout_contains(expected_sha): return False # a newer pull moved HEAD; it owns a fresh obligation # HEAD may sit past ``expected_sha`` by a carried local commit (a cherry-picked hotfix) that no # pull made and no fresh obligation covers; the fleet is held to the code it actually runs, which # is what an equality gate on ``expected_sha`` could never discharge (#119367). target_sha = checkout_sha if not target_sha: return False try: from hermes_cli.update_receipt import collect_fleet_versions, row_is_external fleet = collect_fleet_versions() except Exception as exc: logger.debug("Fleet probe failed; keeping fleet-restart-pending marker: %s", exc) return False if not fleet or (not expected_sha and all(row_is_external(row) for row in fleet)): if owed is not None or not expected_sha: # Absence cannot prove recovery of the recorded inventory / unnamed code; a fleet # whose every row serves ANOTHER checkout root is absence too, not evidence. return False return _discharge_gatewayless_marker(checkout_sha, expected_sha) covered = _fleet_covered_gateways(fleet) if covered is None: return False # unidentified runtime: the matrix cannot vouch for it for row in fleet: if row_is_external(row): continue if row.get("state") != "current" or str(row.get("code_sha")) != target_sha: return False # stale / down / unknown-identity row still owes the restart if owed is not None and not owed <= covered: return False # A gateway this marker owns is absent (down) or unidentifiable. _clear_fleet_restart_pending_marker() logger.debug( "Fleet-restart-pending marker discharged: %d gateway(s) already serve %s", len(fleet), target_sha[:10], ) return True def _receipt_restart_phase_completed(receipt: dict) -> str | None: """Return the pulled SHA when the restart phase completed, even if a later step failed.""" gateway_restart = receipt.get("gateway_restart") if not isinstance(gateway_restart, dict) or not gateway_restart: return None if gateway_restart.get("incomplete") or gateway_restart.get("phase_error"): return None post_sha = (receipt.get("post_update") or {}).get("sha") return str(post_sha) if post_sha else None def _pending_fleet_restart_needed(*, receipt: dict | None = None, pending_manual: list[dict] | None = None) -> bool: """Require identity-matched gateways at checkout HEAD for update catch-up.""" from hermes_cli.update_cmd import _current_checkout_sha from hermes_cli.update_receipt import read_latest_receipt from hermes_cli.update_serve_obligations import retain_receipt_manual_serves if receipt is None: receipt = read_latest_receipt() or {} if pending_manual is None: pending_manual = retain_receipt_manual_serves(receipt) # The HOST obligation owns its inventory; latest.json can belong to an older update. if _fleet_restart_obligation_armed(): return not _marker_only_restart_obsolete() owed = _receipt_owed_gateways(receipt, pending_manual) if not _receipt_reports_stale_runtime(receipt): return False return not _live_fleet_covers_receipt(_current_checkout_sha(), receipt, owed) def _update_owes_fleet_restart(*, receipt: dict | None = None, pending_manual: list[dict] | None = None) -> bool: """Hold a completed restart to the code it pulled, not a later checkout HEAD.""" from hermes_cli.update_cmd import _current_checkout_sha from hermes_cli.update_receipt import read_latest_receipt from hermes_cli.update_serve_obligations import retain_receipt_manual_serves if receipt is None: receipt = read_latest_receipt() or {} if pending_manual is None: pending_manual = retain_receipt_manual_serves(receipt) # A completed older receipt cannot discharge an independent host obligation's inventory. if _fleet_restart_obligation_armed(): return not _marker_only_restart_obsolete() owed = _receipt_owed_gateways(receipt, pending_manual) if not _receipt_reports_stale_runtime(receipt): return False restarted_to = _receipt_restart_phase_completed(receipt) # A fleet an operator has since restarted onto a moved checkout (``hermes gateway restart`` — # the remedy this warning names) has nothing of the update left to owe either. if restarted_to and _live_fleet_covers_receipt(restarted_to, receipt, owed, accept_states=("current", "stale")): return False return not _live_fleet_covers_receipt(_current_checkout_sha(), receipt, owed) def _warn_pending_fleet_restart(*, startup: bool = False) -> None: """Print the specific interrupted-update fleet-restart warning.""" stream = sys.stderr if startup else sys.stdout print("⚠ A previous `hermes update` pulled new code but did not restart running gateways.", file=stream) print(" Gateways may still be serving pre-update modules (mixed sys.modules).", file=stream) if startup: print(" Run `hermes update` or `hermes gateway restart`.", file=stream) def _warn_pending_fleet_restart_on_startup() -> None: """Cheap CLI-startup hint. Never restarts; never raises.""" from hermes_cli.update_receipt import read_latest_receipt from hermes_cli.update_serve_obligations import retain_receipt_manual_serves, warn_pending_manual_serves receipt = read_latest_receipt() or {} pending_manual = None with suppress(Exception): pending_manual = retain_receipt_manual_serves(receipt) with suppress(Exception): if _update_owes_fleet_restart(receipt=receipt, pending_manual=pending_manual): _warn_pending_fleet_restart(startup=True) with suppress(Exception): warn_pending_manual_serves(startup=True, pending_manual=pending_manual) def _systemd_gateway_unit_listings(on_list_timeout=None): """Yield ``(scope, scope_cmd, list-units CompletedProcess)`` per systemd scope that answered. A missing systemctl skips the scope silently; a listing timeout skips it after ``on_list_timeout(scope, exc)`` (when given) so the other scope is still processed. """ for scope, scope_cmd in _SYSTEMD_SCOPES: try: result = _systemctl(scope_cmd + _LIST_GATEWAY_UNITS, timeout=10) except FileNotFoundError: continue except subprocess.TimeoutExpired as exc: if on_list_timeout is not None: on_list_timeout(scope, exc) continue yield scope, scope_cmd, result def _needs_sudo(scope: str) -> bool: return ( scope == "system" and hasattr(os, "geteuid") and os.geteuid() != 0 # windows-footgun: ok — systemd path, Linux-only ) def _unit_main_pid(scope_cmd: list, svc_name: str) -> int: """Live ``MainPID`` of a unit; ``0`` when inactive, unprivileged or unreadable. Property reads need no manage-units privileges, and an unreadable PID is never collapsed: identity that cannot be proved keeps its own restart. """ try: result = _systemctl(list(scope_cmd) + ["show", svc_name, "--property=MainPID", "--value"], timeout=10) except (OSError, subprocess.TimeoutExpired): return 0 if getattr(result, "returncode", 1) != 0: return 0 try: return int((getattr(result, "stdout", "") or "").strip() or 0) except ValueError: return 0 def _restart_systemd_gateway_units_best_effort(failed: list, listings) -> None: """Restart every hermes-gateway/serve unit ONCE PER LIVE HOST PROCESS. One host runs one multiplexing gateway, so leftover per-profile units (``hermes-gateway-.service``) all point at the SAME live ``MainPID``; restarting each in turn restarts the host gateway N times — a self-inflicted N-fold outage triggered by one update. Units that share a live main PID are collapsed to one representative and the others are named as LEGACY units to migrate, never silently dropped. """ from hermes_cli.update_host_obligation import collapse_units_to_host_processes answered = set() targets: dict[str, tuple[str, list, str]] = {} # "/" -> (scope, scope_cmd, unit) for scope, scope_cmd, result in listings: answered.add(scope) if result.returncode != 0: failed.append(f"systemd-{scope} (listing failed)") continue _for_each_systemd_gateway_unit( result.stdout, process_unit=lambda svc_name, _scope=scope, _cmd=scope_cmd: targets.setdefault( f"{_scope}/{svc_name}", (_scope, _cmd, svc_name)), on_unit_timeout=lambda svc_name, exc: failed.append(svc_name), ) keys = list(targets) covered: dict[str, str] = {} if len(keys) > 1: # Only worth a `systemctl show` round when several units could be one process. keys, covered = collapse_units_to_host_processes( keys, lambda key: _unit_main_pid(targets[key][1], targets[key][2])) for unit_key, owner_key in covered.items(): print( f" • {unit_key} is a legacy per-profile unit sharing one host gateway process with " f"{owner_key}; restarting it again would restart that process twice. Fold the units " "together with: hermes gateway migrate" ) for key in keys: scope, scope_cmd, svc_name = targets[key] if not _systemd_unit_owned_by_update(scope_cmd, svc_name): continue manage_cmd = list(scope_cmd) + ["--no-ask-password"] if _needs_sudo(scope): manage_cmd = ["sudo", "-n"] + manage_cmd try: result = _systemctl_reset_and_restart(manage_cmd, svc_name, scope_cmd=scope_cmd) if result.returncode != 0 or not _wait_for_service_active(scope_cmd, svc_name): failed.append(svc_name) except subprocess.TimeoutExpired: failed.append(svc_name) # A timeout or missing executable is not an empty scope. failed.extend(f"systemd-{scope} (listing unavailable)" for scope, _ in _SYSTEMD_SCOPES if scope not in answered) def _live_fleet_current_rows() -> list[dict] | None: """The fleet matrix when the probe finds at least one gateway and every row is ``current`` at the checkout SHA (identity known); ``None`` on any unknown/stale/down row or a failed probe (restart).""" checkout_sha = _current_checkout_sha() if not checkout_sha: return None try: from hermes_cli.update_receipt import collect_fleet_versions fleet = collect_fleet_versions() except Exception as exc: logger.debug("Pending fleet restart: fleet probe failed: %s", exc) return None if not fleet or _fleet_covered_gateways(fleet) is None: return None if all(row.get("state") == "current" and str(row.get("code_sha")) == checkout_sha for row in fleet): return fleet return None def _restart_identity_sha() -> str: """The SHA a completed host restart is stamped with; ``""`` when nothing names the code. ``_current_checkout_sha()`` is ``None`` on every non-git install (zip, pip, Docker), and an empty stamp can never match, so the per-host restart-once guard would be inert exactly on the installs it exists for: each profile's ``hermes update`` would re-kill the one shared multiplexer. The obligation's own ``expected_sha`` — else the receipt's post-update identity — names the same pulled code. """ sha = _current_checkout_sha() if sha: return str(sha) sha = ((_obligation_fields() or {}).get("expected_sha") or "").strip() if sha: return sha with suppress(Exception): from hermes_cli.update_receipt import read_latest_receipt post_update = (read_latest_receipt() or {}).get("post_update") if isinstance(post_update, dict): return str(post_update.get("sha") or "") return "" def _fleet_restart_skip_reason(plan) -> str | None: """Why the completion tail may leave the fleet alone, or ``None`` when a restart is owed. Every route (pulled, already-current, ZIP) now finishes through the same completion tail, so the guards the old catch-up path carried live here: one host runs ONE multiplexing gateway, so a second profile's ``hermes update`` attaches to the restart the first one already stamped (#95294), and a fleet already serving the checkout code (a no-op update, a manual ``hermes gateway restart`` seconds ago) is not re-killed (#117051). The second guard needs BOTH the pre-update plan and the live probe: the live matrix only lists gateways, so a planned ``serve`` still on pre-update code (or any runtime without a stamped identity) keeps the restart — the reconciliation there is what surfaces it. """ from hermes_cli.update_host_obligation import host_restart_already_completed checkout_sha = _restart_identity_sha() if host_restart_already_completed(checkout_sha): return "this host's gateway was already restarted for this update" if (checkout_sha and plan is not None and plan.runtimes and all(str(runtime.code_sha) == checkout_sha for runtime in plan.runtimes) and _live_fleet_current_rows() is not None): return "every running gateway already serves the checkout code" return None def _run_pending_fleet_restart() -> bool: """Historical retry hook; new retries use the ordinary completion owner.""" from hermes_cli._old_updater import stop_for_relaunch stop_for_relaunch(incomplete=True) def _systemctl(cmd: list, *, timeout: float): """Run a systemctl (or sudo systemctl) invocation, capturing utf-8 text with a timeout.""" return subprocess.run(cmd, capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=timeout) # poll() takes signed 32-bit milliseconds; keep headroom for rounding in communicate(). _SYSTEMCTL_RESTART_TIMEOUT_MAX = (2**31 - 1) // 1000 - 1 def _systemd_restart_timeout(scope_cmd: list, svc_name: str, *, start_only: bool = False) -> float: """Outwait the unit's stop + start budgets, not just the systemctl client. A client timeout does not cancel the manager's queued restart. Unknown or infinite limits use systemd's usual 90s per phase so automation stays bounded. Custom ExecStop chains or EXTEND_TIMEOUT_USEC can still exceed this budget; genuine timeouts must continue through the existing per-unit failure path. """ from gateway.shutdown_forensics import parse_systemd_duration_to_us budgets = {"TimeoutStartUSec": 90.0} if not start_only: budgets["TimeoutStopUSec"] = 90.0 try: show = _systemctl( scope_cmd + ["show", svc_name, "--property=TimeoutStopUSec,TimeoutStartUSec"], timeout=5, ) except (FileNotFoundError, subprocess.TimeoutExpired): return sum(budgets.values()) + 15.0 if show.returncode == 0: for line in (show.stdout or "").splitlines(): key, _, raw = line.partition("=") if key in budgets: # The shared parser returns None for infinity/unrecognized units. try: raw = raw.strip() duration = int(raw) if raw.isascii() and raw.isdigit() else parse_systemd_duration_to_us(raw) if duration is not None and duration > 0: budgets[key] = duration / 1_000_000 except (ValueError, OverflowError): pass return min(sum(budgets.values()) + 15.0, _SYSTEMCTL_RESTART_TIMEOUT_MAX) def _systemctl_reset_and_restart(manage_cmd: list, svc_name: str, *, scope_cmd: list | None = None): """``reset-failed`` then ``restart``: a unit parked in failed state by systemd's own auto-restart can wedge a plain ``restart`` against RestartSec backoff and stay dead.""" # Property reads need no manage-units privileges: narrow sudoers may permit # restart/reset-failed but deny show. Keep the same user/system manager scope. timeout = _systemd_restart_timeout(scope_cmd if scope_cmd is not None else manage_cmd, svc_name) _systemctl(manage_cmd + ["reset-failed", svc_name], timeout=10) return _systemctl(manage_cmd + ["restart", svc_name], timeout=timeout) def _systemd_unit_owned_by_update(scope_cmd: list, svc_name: str) -> bool: """Gate a unit restart on the unit's home being one this update owns (#93349). ``hermes-gateway*`` is an account-wide namespace: a second install's ``hermes update`` used to drain and restart the account's real ``hermes-gateway.service`` because the unit was listed, not because it ran the updated code. Foreign or unreadable ownership prints a notice and leaves the unit alone; it is not a failed restart. """ from hermes_cli.update_fleet_scope import describe_skipped_runtime, systemd_unit_hermes_home, home_in_update_scope home = systemd_unit_hermes_home(scope_cmd, svc_name) if home is not None and home_in_update_scope(home): return True print(describe_skipped_runtime("systemd unit", svc_name, home)) return False def _scoped_manual_gateway_pids(pids, *, keep=(), quiet: bool = False) -> list[int]: """*pids* whose live home this update owns (plus *keep*, PIDs already mapped to this install's profile PID files); every other gateway process is named and left running.""" from hermes_cli.update_fleet_scope import describe_skipped_runtime, partition_gateway_pids_by_scope keep = set(keep) owned, foreign = partition_gateway_pids_by_scope([pid for pid in pids if pid not in keep]) if not quiet: for pid, home in foreign: print(describe_skipped_runtime("gateway process", f"PID {pid}", home)) return [pid for pid in pids if pid in keep or pid in owned] def _is_hermes_gateway_unit(unit: str) -> bool: """Exact base unit or hyphenated profile family only: ``startswith("hermes-serve")`` would accept ``hermes-server.service``.""" return ( # list-units is already pattern-filtered, but keep the name gate so a stray non-gateway/serve line # cannot enter the restart path. See #83595. unit == "hermes-gateway.service" or unit.startswith("hermes-gateway-") or unit == "hermes-serve.service" or unit.startswith("hermes-serve-") # #125297: ``hermes-dashboard*`` units are systemd-supervised dashboard backends — the # same fleet this pass restarts. Leaving them out meant a successful update reported # the dashboard ``deferred`` (still on pre-update code) while nothing ever restarted it. or unit == "hermes-dashboard.service" or unit.startswith("hermes-dashboard-") ) def _for_each_systemd_gateway_unit(list_units_stdout: str, *, process_unit, on_unit_timeout) -> None: """Process each hermes-gateway*/hermes-serve* unit from ``systemctl list-units``. ``TimeoutExpired`` from ``process_unit`` is isolated per unit via ``on_unit_timeout`` so one wedged systemctl call cannot abort the rest of the fleet. See #68523. """ for line in (list_units_stdout or "").strip().splitlines(): parts = line.split() if not parts: continue unit = parts[0] if not unit.endswith(".service") or not _is_hermes_gateway_unit(unit): continue svc_name = unit.removesuffix(".service") try: process_unit(svc_name) except subprocess.TimeoutExpired as exc: on_unit_timeout(svc_name, exc) def _service_unit_supports_graceful_sigusr1_restart(svc_name: str) -> bool: """Whether *svc_name* wires SIGUSR1 to a graceful drain-then-restart. Only ``hermes-gateway*`` runs ``gateway/run.py`` (the handler); SIGUSR1 would just kill ``hermes-serve*`` and burn the drain budget, so those go straight to the blunt restart. Same exact/hyphenated shape as ``_for_each_systemd_gateway_unit`` so a near-prefix unit like ``hermes-gatewayd`` is never signalled. See #83438. """ return svc_name == "hermes-gateway" or svc_name.startswith("hermes-gateway-") def _warn_incomplete_gateway_fleet_restart(failed_units: list) -> None: """Print an explicit incomplete-update warning for unrestarted units.""" from hermes_cli.gateway import is_macos if not failed_units: return ordered = list(dict.fromkeys(failed_units)) # de-dup, discovery order print() print("⚠ Update incomplete — some units were not restarted:") for name in ordered: print(f" - {name}") if is_macos(): # A label lands here when launchd wasn't supervising a live process after # the restart — likely deregistered, which `launchctl kickstart` can't revive. # See #88848. print(" Listed services may be deregistered from launchd, or still") print(" running pre-update code (mixed sys.modules). Recover with:") print(" hermes gateway status") print(" launchctl list | grep