fix(update): an in-gateway cron update no longer exits STALE for the gateway it runs inside

`hermes update --yes` from a cron job inside the gateway takes the #100179
ancestor branch (`_drain_or_signal_gateway_for_update` -> self-restart request,
fire-and-forget) — the gateway can only restart after the updater exits. The
post-restart fleet matrix then saw that same gateway on the pre-update code_sha,
printed STALE + "Update not complete" and exited 1 on every nightly run (#119597).

The restart phase now records the pids that accepted a self-restart
(`_GatewayRestartOutcome.self_restart_pending_pids`, threaded from the systemd,
manual and launchd paths) and `collect_fleet_versions(self_restart_pending=)`
turns exactly those rows into `restart_pending` — rendered as "restart pending
(deferred until this process exits)" and excluded from the stale_or_down
verdict. Every other stale/down gateway keeps its verdict; the same pid without
the recorded acceptance is still STALE. The receipt keeps the row's old sha, so
the next `hermes update` / CLI startup hint still verifies the restart landed
via the live fleet (`_receipt_reports_stale_runtime` -> `_live_fleet_covers_receipt`).

Closes #119597
This commit is contained in:
teknium1
2026-09-23 03:18:54 -07:00
committed by Teknium
parent 3a2d86603a
commit 8895b0f119
3 changed files with 182 additions and 18 deletions

View File

@@ -975,7 +975,9 @@ def _warn_incomplete_gateway_fleet_restart(failed_units: list) -> None:
print(" launchctl kickstart -k gui/$UID/<label> # macOS (or user/$UID)")
def _restart_launchd_gateway_after_update(*, supervision_verify: bool = True) -> tuple[list, list]:
def _restart_launchd_gateway_after_update(
*, supervision_verify: bool = True, self_restart_pending: set | None = None,
) -> tuple[list, list]:
"""Restart the invoking profile's launchd gateway after an update.
No ``launchctl list`` gating: a booted-out job (plist present, definition
@@ -994,7 +996,7 @@ def _restart_launchd_gateway_after_update(*, supervision_verify: bool = True) ->
"""
from hermes_cli.gateway import (
get_launchd_label, get_launchd_plist_path, launchd_restart, wait_for_launchd_gateway_supervision,
_launchctl_supervised_pid,
_is_pid_ancestor_of_current_process, _launchctl_supervised_pid,
)
current_label = get_launchd_label()
old_pid = None
@@ -1028,6 +1030,13 @@ def _restart_launchd_gateway_after_update(*, supervision_verify: bool = True) ->
if not supervision_verify:
return [current_label], []
if old_pid is not None and _is_pid_ancestor_of_current_process(old_pid):
# launchd_restart() handed the restart to the gateway this updater runs INSIDE (cron job in
# the gateway tree, #100179): it exits only after this process does, so no fresh supervised
# pid can appear while we wait. Record it as pending for the fleet matrix (#119597).
if self_restart_pending is not None:
self_restart_pending.add(old_pid)
return [current_label], []
# launchd_restart() returning only means "restart REQUESTED" (async). A helper dying
# before first bootstrap, or a bootstrap exiting 0 without registering (macOS 26.6.1),
@@ -1046,6 +1055,7 @@ def _restart_launchd_gateway_after_update(*, supervision_verify: bool = True) ->
def _restart_macos_launchd_gateways(
restarted_services: list, failed_or_stale_units: list, drain_budget: float, *, require_supervision: bool = False,
self_restart_pending: set | None = None,
) -> None:
"""Restart every launchd-managed gateway after an update (macOS).
@@ -1070,7 +1080,8 @@ def _restart_macos_launchd_gateways(
if listing.returncode != 0:
failed_or_stale_units.append("launchd (listing failed)")
return
_restarted, _failed = _restart_launchd_gateway_after_update(supervision_verify=True)
_restarted, _failed = _restart_launchd_gateway_after_update(
supervision_verify=True, self_restart_pending=self_restart_pending)
restarted_services.extend(_restarted)
failed_or_stale_units.extend(_failed)
current_label = get_launchd_label()
@@ -1238,7 +1249,9 @@ def _warn_gateway_restart_phase_aborted(exc: BaseException, pids) -> None:
print(" hermes gateway status")
def _drain_or_signal_gateway_for_update(pid: int, drain_budget: float, label: str) -> bool:
def _drain_or_signal_gateway_for_update(
pid: int, drain_budget: float, label: str, *, self_restart_pending: set | None = None,
) -> bool:
"""Three-way triage (shared by systemd and bare-process paths) for handing a
running gateway over to new code. Returns True when signalled/stopped.
@@ -1246,6 +1259,9 @@ def _drain_or_signal_gateway_for_update(pid: int, drain_budget: float, label: st
tree): waiting is circular (gateway waits on in-flight work → cron session
waits on update → update waits on gateway) and the 1800s force-drain cap burns.
So fire-and-forget: signal restart and return; it completes once THIS process exits.
The pid lands in ``self_restart_pending`` so the fleet matrix can tell "restart
deferred until the updater exits" from "restart never happened" (#119597): the
ancestor is still serving the old code when the matrix runs, by construction.
2. Event loop provably wedged: SIGUSR1 can never drain it; bounded SIGTERM→SIGKILL.
3. Live out-of-tree gateway: graceful SIGUSR1 drain up to ``drain_budget``.
@@ -1263,7 +1279,10 @@ def _drain_or_signal_gateway_for_update(pid: int, drain_budget: float, label: st
"process tree — signalling restart and letting the gateway "
"drain itself (avoids the cron-update deadlock, #100179)"
)
return _request_gateway_self_restart(pid)
accepted = _request_gateway_self_restart(pid)
if accepted and self_restart_pending is not None:
self_restart_pending.add(pid)
return accepted
if probe_gateway_loop_liveness(pid) == GATEWAY_LOOP_WEDGED:
print(f" ⚠ {label}: gateway event loop is unresponsive — skipping drain, forcing a bounded stop...")
_escalate_wedged_gateway(pid)
@@ -1362,7 +1381,7 @@ def _repair_unit_without_fatal_exit_park(svc_name: str, scope: str) -> None:
def _restart_one_systemd_gateway_unit(
svc_name: str, *, scope: str, scope_cmd: list, drain_budget: float, _manage_cmd_cache: dict,
restarted_services: list, failed_or_stale_units: list,
restarted_services: list, failed_or_stale_units: list, self_restart_pending: set | None = None,
) -> None:
"""Restart one active systemd gateway/serve unit: graceful SIGUSR1 drain, then forced restart.
@@ -1390,7 +1409,8 @@ def _restart_one_systemd_gateway_unit(
_main_pid = 0
# Three-way triage (ancestor / wedged / graceful drain).
_graceful_ok = _main_pid > 0 and _drain_or_signal_gateway_for_update(_main_pid, drain_budget, svc_name)
_graceful_ok = _main_pid > 0 and _drain_or_signal_gateway_for_update(
_main_pid, drain_budget, svc_name, self_restart_pending=self_restart_pending)
if _graceful_ok:
# ``Restart=always`` respawns only after RestartSec (60s in our unit; dead time for a
@@ -1466,7 +1486,9 @@ def _restart_one_systemd_gateway_unit(
)
def _restart_systemd_gateway_units(restarted_services, failed_or_stale_units, restarted_scoped_units, drain_budget):
def _restart_systemd_gateway_units(
restarted_services, failed_or_stale_units, restarted_scoped_units, drain_budget, self_restart_pending=None,
):
"""Restart every active hermes-gateway*/hermes-serve* systemd unit (user + system).
Settled units → ``restarted_services`` (bare) and ``restarted_scoped_units``
@@ -1513,6 +1535,7 @@ def _restart_systemd_gateway_units(restarted_services, failed_or_stale_units, re
_manage_cmd_cache=_manage_cmd_cache,
restarted_services=restarted_services,
failed_or_stale_units=failed_or_stale_units,
self_restart_pending=self_restart_pending,
),
on_unit_timeout=_on_unit_timeout,
)
@@ -1540,6 +1563,10 @@ class _GatewayRestartOutcome:
#: ``scope/name`` of every settled systemd unit; the fleet probe stops waiting for a state stamp
#: once none of them is active or activating any more (the successor died, nothing will publish).
restarted_scoped_units: set = field(default_factory=set)
#: Gateways that are ANCESTORS of this updater and accepted a self-restart request
#: (``_drain_or_signal_gateway_for_update`` branch 1): they restart only after this process
#: exits, so the fleet matrix renders them as pending instead of STALE (#119597).
self_restart_pending_pids: set = field(default_factory=set)
def fleet_probe_signals(self) -> tuple:
"""``(pre_restart_pids, killed_pids)`` with the unmapped stops removed — the signals that
@@ -1603,7 +1630,8 @@ def _restart_manual_gateways(out: _GatewayRestartOutcome, _drain_budget) -> None
# SIGUSR1 drain first, SIGTERM fallback if unsupported/over budget — the watcher
# relaunches either way. The helper announces its choice first because a silent
# full-budget wait reads as a hung update.
if not _drain_or_signal_gateway_for_update(pid, _drain_budget, proc.profile):
if not _drain_or_signal_gateway_for_update(
pid, _drain_budget, proc.profile, self_restart_pending=out.self_restart_pending_pids):
with suppress(ProcessLookupError, PermissionError):
os.kill(pid, _signal.SIGTERM)
# Wait ≤5s for exit: Telegram keeps the old getUpdates session ~30s; a new gateway
@@ -1802,13 +1830,17 @@ def _restart_gateway_fleet_after_update(_pre_update_plan, gateway_mode: bool):
out.pre_restart_gateway_pids = None
_restart_systemd_gateway_units(
out.restarted_services, out.failed_or_stale_units, restarted_scoped_units, _drain_budget
out.restarted_services, out.failed_or_stale_units, restarted_scoped_units, _drain_budget,
out.self_restart_pending_pids,
)
# macOS: EVERY ai.hermes.gateway* LaunchAgent (systemd parity).
if is_macos():
with suppress(FileNotFoundError, ImportError):
_restart_macos_launchd_gateways(out.restarted_services, out.failed_or_stale_units, _drain_budget)
_restart_macos_launchd_gateways(
out.restarted_services, out.failed_or_stale_units, _drain_budget,
self_restart_pending=out.self_restart_pending_pids,
)
_restart_manual_gateways(out, _drain_budget)
@@ -1863,19 +1895,21 @@ def _collect_fleet_snapshot(restart, rows_expected: bool) -> list:
``identity_pending`` so the matrix does not call it a pre-stamping gateway.
"""
from hermes_cli.update_receipt import collect_fleet_versions
pending = getattr(restart, "self_restart_pending_pids", None) or None
if not rows_expected:
return collect_fleet_versions(pre_restart_pids=restart.pre_restart_gateway_pids)
return collect_fleet_versions(
pre_restart_pids=restart.pre_restart_gateway_pids, self_restart_pending=pending)
pre_pids = restart.pre_restart_gateway_pids
_fleet_deadline = _time.monotonic() + _FLEET_PROBE_SETTLE_TIMEOUT_SECONDS
while True:
_time.sleep(2.0)
snapshot = collect_fleet_versions(pre_restart_pids=pre_pids)
pending = [row for row in snapshot if _fleet_row_identity_pending(row, pre_pids)]
if snapshot and not pending and not any(row.get("state") == "down" for row in snapshot):
snapshot = collect_fleet_versions(pre_restart_pids=pre_pids, self_restart_pending=pending)
unstamped = [row for row in snapshot if _fleet_row_identity_pending(row, pre_pids)]
if snapshot and not unstamped and not any(row.get("state") == "down" for row in snapshot):
return snapshot
if _time.monotonic() >= _fleet_deadline or _restarted_units_gone(
getattr(restart, "restarted_scoped_units", ())):
for row in pending:
for row in unstamped:
row["identity_pending"] = True
return snapshot

View File

@@ -385,6 +385,9 @@ def _gateway_code_root(pid: int, home: Path) -> Optional[Path]:
EXTERNAL_STATE = "external"
# A gateway this updater runs INSIDE that accepted a self-restart request: it is still on the
# pre-update code by construction and restarts once the updater exits (#100179 / #119597).
RESTART_PENDING_STATE = "restart_pending"
def row_is_external(row: Any) -> bool:
@@ -396,11 +399,14 @@ def _fleet_row(
profile: str, pid: int, code_sha: Any, code_version: Any, expected_sha: Any,
state: str = "unknown", code_root: Optional[Path] = None,
expected_root: Optional[Path] = None, served_profiles: Any = None,
self_restart_pending: Optional[set] = None,
) -> dict[str, Any]:
if state == "unknown" and code_root and expected_root and code_root != expected_root:
state = EXTERNAL_STATE
if state == "unknown" and code_sha and expected_sha:
state = "current" if str(code_sha) == str(expected_sha) else "stale"
if state == "stale" and self_restart_pending and pid in self_restart_pending:
state = RESTART_PENDING_STATE
row = {
"profile": profile, "pid": pid, "code_sha": str(code_sha) if code_sha else None,
"code_version": code_version, "state": state,
@@ -421,9 +427,17 @@ def _fleet_row(
_NOT_EXPECTED_STATES = {"stopped", "startup_failed"}
def collect_fleet_versions(*, pre_restart_pids: Optional[list[int]] = None) -> list[dict[str, Any]]:
def collect_fleet_versions(
*, pre_restart_pids: Optional[list[int]] = None, self_restart_pending: Optional[set] = None,
) -> list[dict[str, Any]]:
"""Snapshot every profile's gateway code identity vs. the current tree.
``self_restart_pending`` — pids of gateways that are ancestors of this updater and accepted a
self-restart request (cron update inside the gateway tree, #100179). They can only restart after
this process exits, so their pre-update ``code_sha`` is expected: such a row is
``restart_pending`` instead of ``stale`` and does not fail the matrix (#119597). Every other
live gateway on the old sha keeps its ``stale`` verdict.
Rollout safety: ``down`` requires membership in ``pre_restart_pids`` — a stale state file from a
long-dead gateway (machine reboot, manual kill weeks ago) must NOT fail every future update.
Without a pre-restart snapshot (``None``/empty) dead PIDs are skipped (historical behavior).
@@ -440,6 +454,7 @@ def collect_fleet_versions(*, pre_restart_pids: Optional[list[int]] = None) -> l
verification gap, #88848/#74973 class).
"""
_pre_restart = {int(p) for p in (pre_restart_pids or []) if isinstance(p, int)}
_pending = {int(p) for p in (self_restart_pending or ()) if isinstance(p, int)}
results: list[dict[str, Any]] = []
expected_sha = _code_identity(refresh=True).get("sha")
expected_root = _updater_code_root()
@@ -458,6 +473,7 @@ def collect_fleet_versions(*, pre_restart_pids: Optional[list[int]] = None) -> l
profile, pid, identity.get("code_sha"), identity.get("code_version"), expected_sha,
served_profiles=identity.get("served_profiles"),
code_root=_gateway_code_root(pid, home), expected_root=expected_root,
self_restart_pending=_pending,
)
results.append({**row, "source": "socket"})
continue
@@ -477,6 +493,7 @@ def collect_fleet_versions(*, pre_restart_pids: Optional[list[int]] = None) -> l
profile, pid, record.get("code_sha"), record.get("code_version"), expected_sha,
served_profiles=record.get("served_profiles"),
code_root=_gateway_code_root(pid, home), expected_root=expected_root,
self_restart_pending=_pending,
)
)
continue
@@ -508,6 +525,10 @@ _FLEET_ROW_LINES = {
"stale": " ✗ {profile} (pid {pid}) @ {short} — STALE (pre-update code)",
"down": " ✗ {profile} — DOWN (gateway was running before the update; pid {pid} is gone and nothing replaced it)",
"external": " ◆ {profile} (pid {pid}) @ {short} — separate checkout, not updated by this run",
RESTART_PENDING_STATE: (
" ↻ {profile} (pid {pid}) @ {short} — restart pending (deferred until this process exits;"
" the update runs inside this gateway)"
),
}
_FLEET_ROW_UNKNOWN = " ? {profile} (pid {pid}) — version unknown (gateway predates version stamping; restart to enable)"
# A gateway pid the pre-update snapshot did not know that had not published its code identity when
@@ -527,7 +548,8 @@ def print_fleet_version_matrix(fleet: list[dict[str, Any]]) -> bool:
provably down (killed by the restart phase, nothing came back), so the caller can escalate.
``unknown`` entries are reported but do NOT fail the update: gateways started before the
code-identity stamp existed have no sha to compare, and failing them would be a false-positive
storm.
storm. ``restart_pending`` entries (the gateway this updater runs inside, self-restart
accepted) are on the old code by construction and do not fail it either (#119597).
"""
if not fleet:
return False
@@ -549,6 +571,10 @@ def print_fleet_version_matrix(fleet: list[dict[str, Any]]) -> bool:
print(" ℹ These profiles run their own checkout and are updated separately:")
for line in external_roots:
print(f" {line}")
if RESTART_PENDING_STATE in states:
print()
print(" ℹ A restart-pending gateway picks up the new code as soon as this update exits;")
print(" verify afterwards with `hermes gateway status`.")
stale_or_down = sum(1 for entry in fleet if entry.get("state") in ("stale", "down"))
if stale_or_down:
print()

View File

@@ -0,0 +1,104 @@
"""A gateway this updater runs INSIDE (cron update in the gateway tree, #100179) accepted a
self-restart request and can only restart after the updater exits — so at fleet-matrix time it is
still on the pre-update code by construction. It used to render STALE and exit 1 on every nightly
cron update (#119597). Now the ancestor pid set recorded by the restart phase turns that row into
``restart_pending``; every OTHER gateway on the old sha keeps its ``stale`` verdict.
"""
import contextlib
import io
import json
import os
import pytest
import hermes_cli.update_cmd_fleet as fleet_mod
import hermes_cli.update_receipt as ur
OLD = "a" * 40
HEAD = "b" * 40
def _fleet_homes(monkeypatch, tmp_path, records: dict[str, dict]) -> None:
"""Default home + named profiles, each with a state file the identity resolver verifies as-is."""
root = tmp_path / ".hermes"
(root / "profiles").mkdir(parents=True)
homes = {"default": root}
for profile, record in records.items():
home = root if profile == "default" else root / "profiles" / profile
home.mkdir(exist_ok=True)
homes[profile] = home
(home / "gateway_state.json").write_text(json.dumps(record), encoding="utf-8")
by_home = {str(home): rec["pid"] for profile, home in homes.items() for rec in [records[profile]]}
monkeypatch.setattr("hermes_cli.build_info.get_code_identity", lambda refresh=False: {"sha": HEAD, "version": "1.0"})
monkeypatch.setattr("hermes_cli.profiles._get_default_hermes_home", lambda: root)
monkeypatch.setattr("hermes_cli.profiles._get_profiles_root", lambda: root / "profiles")
monkeypatch.setattr("gateway.control_socket.identify_gateway", lambda h, **k: None)
monkeypatch.setattr("gateway.status.live_gateway_pid_for_home", lambda h: by_home.get(str(h)))
monkeypatch.setattr(ur, "_gateway_code_root", lambda pid, home: None)
def test_ancestor_with_accepted_self_restart_is_pending_while_other_stale_rows_still_fail(monkeypatch, tmp_path):
ancestor, sibling = os.getpid(), os.getppid()
_fleet_homes(monkeypatch, tmp_path, {
"default": {"pid": ancestor, "gateway_state": "running", "code_sha": OLD},
"ops": {"pid": sibling, "gateway_state": "running", "code_sha": OLD},
})
fleet = ur.collect_fleet_versions(pre_restart_pids=[ancestor, sibling], self_restart_pending={ancestor})
states = {row["profile"]: row["state"] for row in fleet}
assert states == {"default": ur.RESTART_PENDING_STATE, "ops": "stale"}
with contextlib.redirect_stdout(io.StringIO()) as out:
assert ur.print_fleet_version_matrix(fleet) is True # the sibling still fails the update
assert "restart pending" in out.getvalue() and "1 gateway(s) still running the old code" in out.getvalue()
# Without the sibling, the pending row alone is not a failure — and the same pid NOT recorded
# as pending (the restart phase never reached the ancestor branch) is a plain stale verdict.
only_ancestor = [row for row in fleet if row["profile"] == "default"]
with contextlib.redirect_stdout(io.StringIO()):
assert ur.print_fleet_version_matrix(only_ancestor) is False
plain = ur.collect_fleet_versions(pre_restart_pids=[ancestor, sibling])
assert {row["state"] for row in plain} == {"stale"}
def test_restart_phase_records_accepted_self_restart_and_verify_exits_clean(monkeypatch, tmp_path):
"""The ancestor branch of the drain triage feeds the pid set the matrix reads: end to end the
verify phase exits 0 (not 1) and clears the pending marker for an update whose only old-code
gateway is the one it runs inside."""
import hermes_cli.gateway as gateway
import hermes_cli.update_cmd as update_cmd
ancestor = os.getpid()
monkeypatch.setattr(gateway, "_is_pid_ancestor_of_current_process", lambda pid: pid == ancestor)
monkeypatch.setattr(gateway, "_request_gateway_self_restart", lambda pid: True)
pending: set = set()
with contextlib.redirect_stdout(io.StringIO()):
assert fleet_mod._drain_or_signal_gateway_for_update(ancestor, 5.0, "default", self_restart_pending=pending)
assert pending == {ancestor}
_fleet_homes(monkeypatch, tmp_path, {"default": {"pid": ancestor, "gateway_state": "running", "code_sha": OLD}})
monkeypatch.setattr(fleet_mod, "_print_legacy_units_warning", lambda: None)
monkeypatch.setattr(update_cmd, "_finish_dashboard_update_cleanup", lambda *a, **k: None)
monkeypatch.setattr(update_cmd, "_surviving_pre_update_serve_runtimes", lambda plan: [])
monkeypatch.setattr(fleet_mod._time, "sleep", lambda s: None)
cleared = []
monkeypatch.setattr(fleet_mod, "_clear_fleet_restart_pending_marker", lambda: cleared.append(True))
monkeypatch.setattr("hermes_cli.gateway_migrate.maybe_auto_migrate_after_update", lambda: None)
restart = fleet_mod._GatewayRestartOutcome(
incomplete=False, phase_errors=[], pre_restart_gateway_pids=[ancestor], restarted_services=["hermes-gateway"],
failed_or_stale_units=[], relaunched_profiles=[], externally_supervised_profiles=[], killed_pids=set(),
self_restart_pending_pids=pending,
)
with contextlib.redirect_stdout(io.StringIO()) as out:
fleet_mod._verify_fleet_after_update(
restart, _pre_update_plan=None, _windows_gateway_resume=None, node_failures=[], update_complete=True,
) # a SystemExit(1) here is the #119597 symptom
assert "restart pending" in out.getvalue()
assert "Update not complete" not in out.getvalue()
assert cleared == [True]
with contextlib.redirect_stdout(io.StringIO()), pytest.raises(SystemExit) as exc:
restart.self_restart_pending_pids = set() # same fleet, identity not threaded → STALE, exit 1
fleet_mod._verify_fleet_after_update(
restart, _pre_update_plan=None, _windows_gateway_resume=None, node_failures=[], update_complete=True,
)
assert exc.value.code == 1