"""Gateway fleet restart + post-update verification for ``hermes update``. Split out of ``hermes_cli/update_cmd.py``; every name is re-imported there so ``hermes_cli.update_cmd.`` keeps resolving/monkeypatching. Origin helpers are imported lazily inside each function (no import cycle; test patches stay effective). """ import logging from contextlib import suppress import os import subprocess import sys import time as _time from dataclasses import dataclass from pathlib import Path from hermes_cli.update_cmd_common import _best_effort # Log-record parity with the origin module. logger = logging.getLogger("hermes_cli.update_cmd") # Under HERMES_HOME (not next to the venv): records the fleet-restart obligation # after a pull advanced HEAD; cleared only when the restart completes or nothing ran. # The existing ``.update-incomplete`` / ``.lazy-refresh-incomplete`` markers gate dependency/venv repair; # this one is the fleet-restart obligation after a git pull that advanced HEAD (#95294). _FLEET_RESTART_PENDING_NAME = "fleet_restart_pending" _FRESH_RESTART_SUPERVISORS = frozenset({"systemd", "launchd", "service", "s6"}) _SYSTEMD_SCOPES = (("user", ["systemctl", "--user"]), ("system", ["systemctl"])) _LIST_GATEWAY_UNITS = ["list-units", "hermes-gateway*", "hermes-serve*", "--plain", "--no-legend", "--no-pager"] def _write_gateway_update_exit_code(ok: bool) -> None: from hermes_cli.update_cmd import get_hermes_home path = get_hermes_home() / ".update_exit_code" with suppress(OSError): path.write_text("0" if ok else "1", encoding="utf-8") def _fleet_restart_pending_marker_path() -> Path: """HERMES_HOME breadcrumb for a pull that has not yet restarted the fleet.""" from hermes_cli.update_cmd import get_hermes_home return get_hermes_home() / _FLEET_RESTART_PENDING_NAME def _write_fleet_restart_pending_marker(*, expected_sha: str = "") -> None: """Drop the pull→restart obligation breadcrumb. Never raises.""" from hermes_cli.update_cmd import _m path = _fleet_restart_pending_marker_path() if _m()._pytest_owns_live_checkout(path.parent): logger.debug("Skipping fleet-restart-pending marker under pytest (live checkout)") return try: lines = [f"started={_time.time()}", f"pid={os.getpid()}"] if expected_sha: lines.append(f"expected_sha={expected_sha}") path.write_text("\n".join(lines) + "\n", encoding="utf-8") except OSError as exc: logger.debug("Could not write fleet-restart-pending marker: %s", exc) def _clear_fleet_restart_pending_marker() -> None: """Remove the pull→restart obligation breadcrumb. Never raises.""" from hermes_cli.update_cmd import _m _m()._clear_marker_file(_fleet_restart_pending_marker_path(), label="fleet-restart-pending") def _current_checkout_sha() -> str | None: """Current on-disk checkout HEAD, or None if it cannot be resolved.""" from hermes_cli.update_cmd import _capture_head_sha, _m try: from hermes_cli.build_info import get_code_identity sha = (get_code_identity(refresh=True) or {}).get("sha") return str(sha) if sha else None except Exception: return _capture_head_sha(["git"], _m().PROJECT_ROOT) def _receipt_looks_unfinished(receipt: dict) -> bool: """True when *receipt* is from an update that did not finish cleanly.""" gateway_restart = receipt.get("gateway_restart") return bool( receipt.get("stop_reason") or receipt.get("exit_code") not in (0, None) or receipt.get("outcome") in ("failed", "partial", "running") or (isinstance(gateway_restart, dict) and gateway_restart.get("incomplete")) ) def _receipt_reports_stale_runtime(expected_sha: str | None = None) -> bool: """True when ``update_receipts/latest.json`` records a runtime SHA skew. Prefer the post-restart ``fleet`` matrix. ``plan.runtimes[].code_sha`` is captured *before* the pull, so a finished update's plan always looks stale and must not retrigger a restart; consult it only for an unfinished receipt. See #95294. """ from hermes_cli.update_cmd import _current_checkout_sha try: from hermes_cli.update_receipt import read_latest_receipt receipt = read_latest_receipt() except Exception: receipt = None if not isinstance(receipt, dict): return False expected_sha = expected_sha or _current_checkout_sha() if not expected_sha: return False def _sha_mismatch(code_sha) -> bool: return bool(code_sha) and str(code_sha) != str(expected_sha) fleet = receipt.get("fleet") if isinstance(fleet, list) and fleet: return any( isinstance(entry, dict) and (entry.get("state") == "stale" or _sha_mismatch(entry.get("code_sha"))) for entry in fleet ) if not _receipt_looks_unfinished(receipt): return False plan = receipt.get("plan") if not isinstance(plan, dict): return False return any( isinstance(runtime, dict) and _sha_mismatch(runtime.get("code_sha")) for runtime in plan.get("runtimes") or [] ) def _pending_fleet_restart_needed() -> bool: """True when a prior pull still owes the fleet a restart. See #95294. """ with suppress(OSError): if _fleet_restart_pending_marker_path().is_file(): return True return _receipt_reports_stale_runtime() def _warn_pending_fleet_restart(*, startup: bool = False) -> None: """Print the specific interrupted-update fleet-restart warning.""" stream = sys.stderr if startup else sys.stdout print("⚠ A previous `hermes update` pulled new code but did not restart running gateways.", file=stream) print(" Gateways may still be serving pre-update modules (mixed sys.modules).", file=stream) if startup: print(" Run `hermes update` or `hermes gateway restart`.", file=stream) def _warn_pending_fleet_restart_on_startup() -> None: """Cheap CLI-startup hint. Never restarts; never raises.""" with suppress(Exception): if _pending_fleet_restart_needed(): _warn_pending_fleet_restart(startup=True) def _systemd_gateway_unit_listings(on_list_timeout=None): """Yield ``(scope, scope_cmd, list-units CompletedProcess)`` per systemd scope that answered. A missing systemctl skips the scope silently; a listing timeout skips it after ``on_list_timeout(scope, exc)`` (when given) so the other scope is still processed. """ for scope, scope_cmd in _SYSTEMD_SCOPES: try: result = _systemctl(scope_cmd + _LIST_GATEWAY_UNITS, timeout=10) except FileNotFoundError: continue except subprocess.TimeoutExpired as exc: if on_list_timeout is not None: on_list_timeout(scope, exc) continue yield scope, scope_cmd, result def _needs_sudo(scope: str) -> bool: return ( scope == "system" and hasattr(os, "geteuid") and os.geteuid() != 0 # windows-footgun: ok — systemd path, Linux-only ) def _restart_systemd_gateway_units_best_effort(failed: list) -> None: """Best-effort ``systemctl restart`` of every hermes-gateway/serve unit.""" for scope, scope_cmd, result in _systemd_gateway_unit_listings(): if result.returncode != 0: continue def process_unit(svc_name: str, _scope=scope, _cmd=scope_cmd) -> None: restart_cmd = list(_cmd) + ["--no-ask-password", "restart", svc_name] if _needs_sudo(_scope): restart_cmd = ["sudo", "-n"] + restart_cmd _systemctl(restart_cmd, timeout=30) _for_each_systemd_gateway_unit( result.stdout, process_unit=process_unit, on_unit_timeout=lambda svc_name, exc: failed.append(svc_name), ) def _run_pending_fleet_restart() -> bool: """Catch-up restart for gateways left on pre-update code. Never raises. True when the restart completed or nothing was running; False if incomplete. See #95294. """ from hermes_cli.update_cmd import _m print("→ Restarting gateways left on pre-update code...") with suppress(Exception): _m()._purge_stale_hermes_modules() # Warn if legacy Hermes gateway unit files are still installed. When both hermes.service (from a # pre-rename install) and the current hermes-gateway.service are enabled, they SIGTERM-fight for the # same bot token (see PR #11909). Flagging here means every `hermes update` surfaces the issue until the # user migrates. try: from hermes_cli.gateway import ( find_gateway_pids, is_macos, is_windows, kill_gateway_processes, supports_systemd_services, _wait_for_gateway_exit, ) except Exception as exc: _warn_gateway_restart_phase_aborted(exc, None) return False try: pids = list(find_gateway_pids(all_profiles=True)) except Exception as exc: logger.debug("Pending fleet restart: gateway probe failed: %s", exc) pids = None if pids == []: print(" ✓ No running gateways — nothing to restart.") return True failed: list = [] try: # --- Systemd services (Linux) --- Discover all hermes-gateway* units (default + profiles) plus # hermes-serve* units (the Desktop app's backend, #83438). if supports_systemd_services(): _restart_systemd_gateway_units_best_effort(failed) # --- Launchd services (macOS) --- Restart EVERY ai.hermes.gateway* LaunchAgent, not only the # invoking profile's — parity with the systemd branch above (#41403). Per-label TimeoutExpired # isolation happens inside. if is_macos(): try: _restart_macos_launchd_gateways([], failed, 45.0) except Exception as exc: logger.debug("Pending fleet restart: launchd failed: %s", exc) failed.append("launchd") if is_windows(): try: from hermes_cli import gateway_windows if gateway_windows.is_installed(): gateway_windows.restart() except Exception as exc: logger.debug("Pending fleet restart: Windows failed: %s", exc) failed.append("windows-gateway") try: leftover = list(find_gateway_pids(all_profiles=True)) except Exception: leftover = list(pids or []) if leftover: with _best_effort('Pending fleet restart: PID stop failed: %s'): kill_gateway_processes(all_profiles=True) _wait_for_gateway_exit(timeout=5.0, force_after=None) if failed: _warn_incomplete_gateway_fleet_restart(failed) return False print(" ✓ Pending fleet restart completed.") return True except Exception as exc: try: surviving = list(find_gateway_pids(all_profiles=True)) except Exception: surviving = pids _warn_gateway_restart_phase_aborted(exc, surviving) return False def _apply_pending_fleet_restart_catchup() -> None: """On an already-up-to-date ``hermes update``, finish a skipped restart. No-op when nothing is pending; exits 1 on incomplete catch-up so automation does not treat the fleet as healthy. """ from hermes_cli.update_cmd import _run_pending_fleet_restart if not _pending_fleet_restart_needed(): return print() _warn_pending_fleet_restart() print("→ Running the pending fleet restart...") if _run_pending_fleet_restart(): _clear_fleet_restart_pending_marker() return print(" ⚠ Fleet restart incomplete. Recover with: hermes gateway restart") sys.exit(1) def _systemctl(cmd: list, *, timeout: float): """Run a systemctl (or sudo systemctl) invocation, capturing utf-8 text with a timeout.""" return subprocess.run(cmd, capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=timeout) def _systemctl_reset_and_restart(manage_cmd: list, svc_name: str): """``reset-failed`` then ``restart``: a unit parked in failed state by systemd's own auto-restart can wedge a plain ``restart`` against RestartSec backoff and stay dead.""" _systemctl(manage_cmd + ["reset-failed", svc_name], timeout=10) return _systemctl(manage_cmd + ["restart", svc_name], timeout=15) def _is_hermes_gateway_unit(unit: str) -> bool: """Exact base unit or hyphenated profile family only: ``startswith("hermes-serve")`` would accept ``hermes-server.service``.""" return ( # list-units is already pattern-filtered, but keep the name gate so a stray non-gateway/serve line # cannot enter the restart path. See #83595. unit == "hermes-gateway.service" or unit.startswith("hermes-gateway-") or unit == "hermes-serve.service" or unit.startswith("hermes-serve-") ) def _for_each_systemd_gateway_unit(list_units_stdout: str, *, process_unit, on_unit_timeout) -> None: """Process each hermes-gateway*/hermes-serve* unit from ``systemctl list-units``. ``TimeoutExpired`` from ``process_unit`` is isolated per unit via ``on_unit_timeout`` so one wedged systemctl call cannot abort the rest of the fleet. See #68523. """ for line in (list_units_stdout or "").strip().splitlines(): parts = line.split() if not parts: continue unit = parts[0] if not unit.endswith(".service") or not _is_hermes_gateway_unit(unit): continue svc_name = unit.removesuffix(".service") try: process_unit(svc_name) except subprocess.TimeoutExpired as exc: on_unit_timeout(svc_name, exc) def _service_unit_supports_graceful_sigusr1_restart(svc_name: str) -> bool: """Whether *svc_name* wires SIGUSR1 to a graceful drain-then-restart. Only ``hermes-gateway*`` runs ``gateway/run.py`` (the handler); SIGUSR1 would just kill ``hermes-serve*`` and burn the drain budget, so those go straight to the blunt restart. Same exact/hyphenated shape as ``_for_each_systemd_gateway_unit`` so a near-prefix unit like ``hermes-gatewayd`` is never signalled. See #83438. """ return svc_name == "hermes-gateway" or svc_name.startswith("hermes-gateway-") def _warn_incomplete_gateway_fleet_restart(failed_units: list) -> None: """Print an explicit incomplete-update warning for unrestarted units.""" from hermes_cli.gateway import is_macos if not failed_units: return ordered = list(dict.fromkeys(failed_units)) # de-dup, discovery order print() print("⚠ Update incomplete — some units were not restarted:") for name in ordered: print(f" - {name}") if is_macos(): # A label lands here when launchd wasn't supervising a live process after # the restart — likely deregistered, which `launchctl kickstart` can't revive. # See #88848. print(" Listed services may be deregistered from launchd, or still") print(" running pre-update code (mixed sys.modules). Recover with:") print(" hermes gateway status") print(" launchctl list | grep