fix(update): an in-gateway cron update no longer exits STALE for the gateway it runs inside
`hermes update --yes` from a cron job inside the gateway takes the #100179 ancestor branch (`_drain_or_signal_gateway_for_update` -> self-restart request, fire-and-forget) — the gateway can only restart after the updater exits. The post-restart fleet matrix then saw that same gateway on the pre-update code_sha, printed STALE + "Update not complete" and exited 1 on every nightly run (#119597). The restart phase now records the pids that accepted a self-restart (`_GatewayRestartOutcome.self_restart_pending_pids`, threaded from the systemd, manual and launchd paths) and `collect_fleet_versions(self_restart_pending=)` turns exactly those rows into `restart_pending` — rendered as "restart pending (deferred until this process exits)" and excluded from the stale_or_down verdict. Every other stale/down gateway keeps its verdict; the same pid without the recorded acceptance is still STALE. The receipt keeps the row's old sha, so the next `hermes update` / CLI startup hint still verifies the restart landed via the live fleet (`_receipt_reports_stale_runtime` -> `_live_fleet_covers_receipt`). Closes #119597
This commit is contained in:
@@ -975,7 +975,9 @@ def _warn_incomplete_gateway_fleet_restart(failed_units: list) -> None:
|
||||
print(" launchctl kickstart -k gui/$UID/<label> # macOS (or user/$UID)")
|
||||
|
||||
|
||||
def _restart_launchd_gateway_after_update(*, supervision_verify: bool = True) -> tuple[list, list]:
|
||||
def _restart_launchd_gateway_after_update(
|
||||
*, supervision_verify: bool = True, self_restart_pending: set | None = None,
|
||||
) -> tuple[list, list]:
|
||||
"""Restart the invoking profile's launchd gateway after an update.
|
||||
|
||||
No ``launchctl list`` gating: a booted-out job (plist present, definition
|
||||
@@ -994,7 +996,7 @@ def _restart_launchd_gateway_after_update(*, supervision_verify: bool = True) ->
|
||||
"""
|
||||
from hermes_cli.gateway import (
|
||||
get_launchd_label, get_launchd_plist_path, launchd_restart, wait_for_launchd_gateway_supervision,
|
||||
_launchctl_supervised_pid,
|
||||
_is_pid_ancestor_of_current_process, _launchctl_supervised_pid,
|
||||
)
|
||||
current_label = get_launchd_label()
|
||||
old_pid = None
|
||||
@@ -1028,6 +1030,13 @@ def _restart_launchd_gateway_after_update(*, supervision_verify: bool = True) ->
|
||||
|
||||
if not supervision_verify:
|
||||
return [current_label], []
|
||||
if old_pid is not None and _is_pid_ancestor_of_current_process(old_pid):
|
||||
# launchd_restart() handed the restart to the gateway this updater runs INSIDE (cron job in
|
||||
# the gateway tree, #100179): it exits only after this process does, so no fresh supervised
|
||||
# pid can appear while we wait. Record it as pending for the fleet matrix (#119597).
|
||||
if self_restart_pending is not None:
|
||||
self_restart_pending.add(old_pid)
|
||||
return [current_label], []
|
||||
|
||||
# launchd_restart() returning only means "restart REQUESTED" (async). A helper dying
|
||||
# before first bootstrap, or a bootstrap exiting 0 without registering (macOS 26.6.1),
|
||||
@@ -1046,6 +1055,7 @@ def _restart_launchd_gateway_after_update(*, supervision_verify: bool = True) ->
|
||||
|
||||
def _restart_macos_launchd_gateways(
|
||||
restarted_services: list, failed_or_stale_units: list, drain_budget: float, *, require_supervision: bool = False,
|
||||
self_restart_pending: set | None = None,
|
||||
) -> None:
|
||||
"""Restart every launchd-managed gateway after an update (macOS).
|
||||
|
||||
@@ -1070,7 +1080,8 @@ def _restart_macos_launchd_gateways(
|
||||
if listing.returncode != 0:
|
||||
failed_or_stale_units.append("launchd (listing failed)")
|
||||
return
|
||||
_restarted, _failed = _restart_launchd_gateway_after_update(supervision_verify=True)
|
||||
_restarted, _failed = _restart_launchd_gateway_after_update(
|
||||
supervision_verify=True, self_restart_pending=self_restart_pending)
|
||||
restarted_services.extend(_restarted)
|
||||
failed_or_stale_units.extend(_failed)
|
||||
current_label = get_launchd_label()
|
||||
@@ -1238,7 +1249,9 @@ def _warn_gateway_restart_phase_aborted(exc: BaseException, pids) -> None:
|
||||
print(" hermes gateway status")
|
||||
|
||||
|
||||
def _drain_or_signal_gateway_for_update(pid: int, drain_budget: float, label: str) -> bool:
|
||||
def _drain_or_signal_gateway_for_update(
|
||||
pid: int, drain_budget: float, label: str, *, self_restart_pending: set | None = None,
|
||||
) -> bool:
|
||||
"""Three-way triage (shared by systemd and bare-process paths) for handing a
|
||||
running gateway over to new code. Returns True when signalled/stopped.
|
||||
|
||||
@@ -1246,6 +1259,9 @@ def _drain_or_signal_gateway_for_update(pid: int, drain_budget: float, label: st
|
||||
tree): waiting is circular (gateway waits on in-flight work → cron session
|
||||
waits on update → update waits on gateway) and the 1800s force-drain cap burns.
|
||||
So fire-and-forget: signal restart and return; it completes once THIS process exits.
|
||||
The pid lands in ``self_restart_pending`` so the fleet matrix can tell "restart
|
||||
deferred until the updater exits" from "restart never happened" (#119597): the
|
||||
ancestor is still serving the old code when the matrix runs, by construction.
|
||||
2. Event loop provably wedged: SIGUSR1 can never drain it; bounded SIGTERM→SIGKILL.
|
||||
3. Live out-of-tree gateway: graceful SIGUSR1 drain up to ``drain_budget``.
|
||||
|
||||
@@ -1263,7 +1279,10 @@ def _drain_or_signal_gateway_for_update(pid: int, drain_budget: float, label: st
|
||||
"process tree — signalling restart and letting the gateway "
|
||||
"drain itself (avoids the cron-update deadlock, #100179)"
|
||||
)
|
||||
return _request_gateway_self_restart(pid)
|
||||
accepted = _request_gateway_self_restart(pid)
|
||||
if accepted and self_restart_pending is not None:
|
||||
self_restart_pending.add(pid)
|
||||
return accepted
|
||||
if probe_gateway_loop_liveness(pid) == GATEWAY_LOOP_WEDGED:
|
||||
print(f" ⚠ {label}: gateway event loop is unresponsive — skipping drain, forcing a bounded stop...")
|
||||
_escalate_wedged_gateway(pid)
|
||||
@@ -1362,7 +1381,7 @@ def _repair_unit_without_fatal_exit_park(svc_name: str, scope: str) -> None:
|
||||
|
||||
def _restart_one_systemd_gateway_unit(
|
||||
svc_name: str, *, scope: str, scope_cmd: list, drain_budget: float, _manage_cmd_cache: dict,
|
||||
restarted_services: list, failed_or_stale_units: list,
|
||||
restarted_services: list, failed_or_stale_units: list, self_restart_pending: set | None = None,
|
||||
) -> None:
|
||||
"""Restart one active systemd gateway/serve unit: graceful SIGUSR1 drain, then forced restart.
|
||||
|
||||
@@ -1390,7 +1409,8 @@ def _restart_one_systemd_gateway_unit(
|
||||
_main_pid = 0
|
||||
|
||||
# Three-way triage (ancestor / wedged / graceful drain).
|
||||
_graceful_ok = _main_pid > 0 and _drain_or_signal_gateway_for_update(_main_pid, drain_budget, svc_name)
|
||||
_graceful_ok = _main_pid > 0 and _drain_or_signal_gateway_for_update(
|
||||
_main_pid, drain_budget, svc_name, self_restart_pending=self_restart_pending)
|
||||
|
||||
if _graceful_ok:
|
||||
# ``Restart=always`` respawns only after RestartSec (60s in our unit; dead time for a
|
||||
@@ -1466,7 +1486,9 @@ def _restart_one_systemd_gateway_unit(
|
||||
)
|
||||
|
||||
|
||||
def _restart_systemd_gateway_units(restarted_services, failed_or_stale_units, restarted_scoped_units, drain_budget):
|
||||
def _restart_systemd_gateway_units(
|
||||
restarted_services, failed_or_stale_units, restarted_scoped_units, drain_budget, self_restart_pending=None,
|
||||
):
|
||||
"""Restart every active hermes-gateway*/hermes-serve* systemd unit (user + system).
|
||||
|
||||
Settled units → ``restarted_services`` (bare) and ``restarted_scoped_units``
|
||||
@@ -1513,6 +1535,7 @@ def _restart_systemd_gateway_units(restarted_services, failed_or_stale_units, re
|
||||
_manage_cmd_cache=_manage_cmd_cache,
|
||||
restarted_services=restarted_services,
|
||||
failed_or_stale_units=failed_or_stale_units,
|
||||
self_restart_pending=self_restart_pending,
|
||||
),
|
||||
on_unit_timeout=_on_unit_timeout,
|
||||
)
|
||||
@@ -1540,6 +1563,10 @@ class _GatewayRestartOutcome:
|
||||
#: ``scope/name`` of every settled systemd unit; the fleet probe stops waiting for a state stamp
|
||||
#: once none of them is active or activating any more (the successor died, nothing will publish).
|
||||
restarted_scoped_units: set = field(default_factory=set)
|
||||
#: Gateways that are ANCESTORS of this updater and accepted a self-restart request
|
||||
#: (``_drain_or_signal_gateway_for_update`` branch 1): they restart only after this process
|
||||
#: exits, so the fleet matrix renders them as pending instead of STALE (#119597).
|
||||
self_restart_pending_pids: set = field(default_factory=set)
|
||||
|
||||
def fleet_probe_signals(self) -> tuple:
|
||||
"""``(pre_restart_pids, killed_pids)`` with the unmapped stops removed — the signals that
|
||||
@@ -1603,7 +1630,8 @@ def _restart_manual_gateways(out: _GatewayRestartOutcome, _drain_budget) -> None
|
||||
# SIGUSR1 drain first, SIGTERM fallback if unsupported/over budget — the watcher
|
||||
# relaunches either way. The helper announces its choice first because a silent
|
||||
# full-budget wait reads as a hung update.
|
||||
if not _drain_or_signal_gateway_for_update(pid, _drain_budget, proc.profile):
|
||||
if not _drain_or_signal_gateway_for_update(
|
||||
pid, _drain_budget, proc.profile, self_restart_pending=out.self_restart_pending_pids):
|
||||
with suppress(ProcessLookupError, PermissionError):
|
||||
os.kill(pid, _signal.SIGTERM)
|
||||
# Wait ≤5s for exit: Telegram keeps the old getUpdates session ~30s; a new gateway
|
||||
@@ -1802,13 +1830,17 @@ def _restart_gateway_fleet_after_update(_pre_update_plan, gateway_mode: bool):
|
||||
out.pre_restart_gateway_pids = None
|
||||
|
||||
_restart_systemd_gateway_units(
|
||||
out.restarted_services, out.failed_or_stale_units, restarted_scoped_units, _drain_budget
|
||||
out.restarted_services, out.failed_or_stale_units, restarted_scoped_units, _drain_budget,
|
||||
out.self_restart_pending_pids,
|
||||
)
|
||||
|
||||
# macOS: EVERY ai.hermes.gateway* LaunchAgent (systemd parity).
|
||||
if is_macos():
|
||||
with suppress(FileNotFoundError, ImportError):
|
||||
_restart_macos_launchd_gateways(out.restarted_services, out.failed_or_stale_units, _drain_budget)
|
||||
_restart_macos_launchd_gateways(
|
||||
out.restarted_services, out.failed_or_stale_units, _drain_budget,
|
||||
self_restart_pending=out.self_restart_pending_pids,
|
||||
)
|
||||
|
||||
_restart_manual_gateways(out, _drain_budget)
|
||||
|
||||
@@ -1863,19 +1895,21 @@ def _collect_fleet_snapshot(restart, rows_expected: bool) -> list:
|
||||
``identity_pending`` so the matrix does not call it a pre-stamping gateway.
|
||||
"""
|
||||
from hermes_cli.update_receipt import collect_fleet_versions
|
||||
pending = getattr(restart, "self_restart_pending_pids", None) or None
|
||||
if not rows_expected:
|
||||
return collect_fleet_versions(pre_restart_pids=restart.pre_restart_gateway_pids)
|
||||
return collect_fleet_versions(
|
||||
pre_restart_pids=restart.pre_restart_gateway_pids, self_restart_pending=pending)
|
||||
pre_pids = restart.pre_restart_gateway_pids
|
||||
_fleet_deadline = _time.monotonic() + _FLEET_PROBE_SETTLE_TIMEOUT_SECONDS
|
||||
while True:
|
||||
_time.sleep(2.0)
|
||||
snapshot = collect_fleet_versions(pre_restart_pids=pre_pids)
|
||||
pending = [row for row in snapshot if _fleet_row_identity_pending(row, pre_pids)]
|
||||
if snapshot and not pending and not any(row.get("state") == "down" for row in snapshot):
|
||||
snapshot = collect_fleet_versions(pre_restart_pids=pre_pids, self_restart_pending=pending)
|
||||
unstamped = [row for row in snapshot if _fleet_row_identity_pending(row, pre_pids)]
|
||||
if snapshot and not unstamped and not any(row.get("state") == "down" for row in snapshot):
|
||||
return snapshot
|
||||
if _time.monotonic() >= _fleet_deadline or _restarted_units_gone(
|
||||
getattr(restart, "restarted_scoped_units", ())):
|
||||
for row in pending:
|
||||
for row in unstamped:
|
||||
row["identity_pending"] = True
|
||||
return snapshot
|
||||
|
||||
|
||||
@@ -385,6 +385,9 @@ def _gateway_code_root(pid: int, home: Path) -> Optional[Path]:
|
||||
|
||||
|
||||
EXTERNAL_STATE = "external"
|
||||
# A gateway this updater runs INSIDE that accepted a self-restart request: it is still on the
|
||||
# pre-update code by construction and restarts once the updater exits (#100179 / #119597).
|
||||
RESTART_PENDING_STATE = "restart_pending"
|
||||
|
||||
|
||||
def row_is_external(row: Any) -> bool:
|
||||
@@ -396,11 +399,14 @@ def _fleet_row(
|
||||
profile: str, pid: int, code_sha: Any, code_version: Any, expected_sha: Any,
|
||||
state: str = "unknown", code_root: Optional[Path] = None,
|
||||
expected_root: Optional[Path] = None, served_profiles: Any = None,
|
||||
self_restart_pending: Optional[set] = None,
|
||||
) -> dict[str, Any]:
|
||||
if state == "unknown" and code_root and expected_root and code_root != expected_root:
|
||||
state = EXTERNAL_STATE
|
||||
if state == "unknown" and code_sha and expected_sha:
|
||||
state = "current" if str(code_sha) == str(expected_sha) else "stale"
|
||||
if state == "stale" and self_restart_pending and pid in self_restart_pending:
|
||||
state = RESTART_PENDING_STATE
|
||||
row = {
|
||||
"profile": profile, "pid": pid, "code_sha": str(code_sha) if code_sha else None,
|
||||
"code_version": code_version, "state": state,
|
||||
@@ -421,9 +427,17 @@ def _fleet_row(
|
||||
_NOT_EXPECTED_STATES = {"stopped", "startup_failed"}
|
||||
|
||||
|
||||
def collect_fleet_versions(*, pre_restart_pids: Optional[list[int]] = None) -> list[dict[str, Any]]:
|
||||
def collect_fleet_versions(
|
||||
*, pre_restart_pids: Optional[list[int]] = None, self_restart_pending: Optional[set] = None,
|
||||
) -> list[dict[str, Any]]:
|
||||
"""Snapshot every profile's gateway code identity vs. the current tree.
|
||||
|
||||
``self_restart_pending`` — pids of gateways that are ancestors of this updater and accepted a
|
||||
self-restart request (cron update inside the gateway tree, #100179). They can only restart after
|
||||
this process exits, so their pre-update ``code_sha`` is expected: such a row is
|
||||
``restart_pending`` instead of ``stale`` and does not fail the matrix (#119597). Every other
|
||||
live gateway on the old sha keeps its ``stale`` verdict.
|
||||
|
||||
Rollout safety: ``down`` requires membership in ``pre_restart_pids`` — a stale state file from a
|
||||
long-dead gateway (machine reboot, manual kill weeks ago) must NOT fail every future update.
|
||||
Without a pre-restart snapshot (``None``/empty) dead PIDs are skipped (historical behavior).
|
||||
@@ -440,6 +454,7 @@ def collect_fleet_versions(*, pre_restart_pids: Optional[list[int]] = None) -> l
|
||||
verification gap, #88848/#74973 class).
|
||||
"""
|
||||
_pre_restart = {int(p) for p in (pre_restart_pids or []) if isinstance(p, int)}
|
||||
_pending = {int(p) for p in (self_restart_pending or ()) if isinstance(p, int)}
|
||||
results: list[dict[str, Any]] = []
|
||||
expected_sha = _code_identity(refresh=True).get("sha")
|
||||
expected_root = _updater_code_root()
|
||||
@@ -458,6 +473,7 @@ def collect_fleet_versions(*, pre_restart_pids: Optional[list[int]] = None) -> l
|
||||
profile, pid, identity.get("code_sha"), identity.get("code_version"), expected_sha,
|
||||
served_profiles=identity.get("served_profiles"),
|
||||
code_root=_gateway_code_root(pid, home), expected_root=expected_root,
|
||||
self_restart_pending=_pending,
|
||||
)
|
||||
results.append({**row, "source": "socket"})
|
||||
continue
|
||||
@@ -477,6 +493,7 @@ def collect_fleet_versions(*, pre_restart_pids: Optional[list[int]] = None) -> l
|
||||
profile, pid, record.get("code_sha"), record.get("code_version"), expected_sha,
|
||||
served_profiles=record.get("served_profiles"),
|
||||
code_root=_gateway_code_root(pid, home), expected_root=expected_root,
|
||||
self_restart_pending=_pending,
|
||||
)
|
||||
)
|
||||
continue
|
||||
@@ -508,6 +525,10 @@ _FLEET_ROW_LINES = {
|
||||
"stale": " ✗ {profile} (pid {pid}) @ {short} — STALE (pre-update code)",
|
||||
"down": " ✗ {profile} — DOWN (gateway was running before the update; pid {pid} is gone and nothing replaced it)",
|
||||
"external": " ◆ {profile} (pid {pid}) @ {short} — separate checkout, not updated by this run",
|
||||
RESTART_PENDING_STATE: (
|
||||
" ↻ {profile} (pid {pid}) @ {short} — restart pending (deferred until this process exits;"
|
||||
" the update runs inside this gateway)"
|
||||
),
|
||||
}
|
||||
_FLEET_ROW_UNKNOWN = " ? {profile} (pid {pid}) — version unknown (gateway predates version stamping; restart to enable)"
|
||||
# A gateway pid the pre-update snapshot did not know that had not published its code identity when
|
||||
@@ -527,7 +548,8 @@ def print_fleet_version_matrix(fleet: list[dict[str, Any]]) -> bool:
|
||||
provably down (killed by the restart phase, nothing came back), so the caller can escalate.
|
||||
``unknown`` entries are reported but do NOT fail the update: gateways started before the
|
||||
code-identity stamp existed have no sha to compare, and failing them would be a false-positive
|
||||
storm.
|
||||
storm. ``restart_pending`` entries (the gateway this updater runs inside, self-restart
|
||||
accepted) are on the old code by construction and do not fail it either (#119597).
|
||||
"""
|
||||
if not fleet:
|
||||
return False
|
||||
@@ -549,6 +571,10 @@ def print_fleet_version_matrix(fleet: list[dict[str, Any]]) -> bool:
|
||||
print(" ℹ These profiles run their own checkout and are updated separately:")
|
||||
for line in external_roots:
|
||||
print(f" {line}")
|
||||
if RESTART_PENDING_STATE in states:
|
||||
print()
|
||||
print(" ℹ A restart-pending gateway picks up the new code as soon as this update exits;")
|
||||
print(" verify afterwards with `hermes gateway status`.")
|
||||
stale_or_down = sum(1 for entry in fleet if entry.get("state") in ("stale", "down"))
|
||||
if stale_or_down:
|
||||
print()
|
||||
|
||||
104
tests/hermes_cli/test_fleet_matrix_self_restart_pending.py
Normal file
104
tests/hermes_cli/test_fleet_matrix_self_restart_pending.py
Normal file
@@ -0,0 +1,104 @@
|
||||
"""A gateway this updater runs INSIDE (cron update in the gateway tree, #100179) accepted a
|
||||
self-restart request and can only restart after the updater exits — so at fleet-matrix time it is
|
||||
still on the pre-update code by construction. It used to render STALE and exit 1 on every nightly
|
||||
cron update (#119597). Now the ancestor pid set recorded by the restart phase turns that row into
|
||||
``restart_pending``; every OTHER gateway on the old sha keeps its ``stale`` verdict.
|
||||
"""
|
||||
|
||||
import contextlib
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
||||
import hermes_cli.update_cmd_fleet as fleet_mod
|
||||
import hermes_cli.update_receipt as ur
|
||||
|
||||
OLD = "a" * 40
|
||||
HEAD = "b" * 40
|
||||
|
||||
|
||||
def _fleet_homes(monkeypatch, tmp_path, records: dict[str, dict]) -> None:
|
||||
"""Default home + named profiles, each with a state file the identity resolver verifies as-is."""
|
||||
root = tmp_path / ".hermes"
|
||||
(root / "profiles").mkdir(parents=True)
|
||||
homes = {"default": root}
|
||||
for profile, record in records.items():
|
||||
home = root if profile == "default" else root / "profiles" / profile
|
||||
home.mkdir(exist_ok=True)
|
||||
homes[profile] = home
|
||||
(home / "gateway_state.json").write_text(json.dumps(record), encoding="utf-8")
|
||||
by_home = {str(home): rec["pid"] for profile, home in homes.items() for rec in [records[profile]]}
|
||||
monkeypatch.setattr("hermes_cli.build_info.get_code_identity", lambda refresh=False: {"sha": HEAD, "version": "1.0"})
|
||||
monkeypatch.setattr("hermes_cli.profiles._get_default_hermes_home", lambda: root)
|
||||
monkeypatch.setattr("hermes_cli.profiles._get_profiles_root", lambda: root / "profiles")
|
||||
monkeypatch.setattr("gateway.control_socket.identify_gateway", lambda h, **k: None)
|
||||
monkeypatch.setattr("gateway.status.live_gateway_pid_for_home", lambda h: by_home.get(str(h)))
|
||||
monkeypatch.setattr(ur, "_gateway_code_root", lambda pid, home: None)
|
||||
|
||||
|
||||
def test_ancestor_with_accepted_self_restart_is_pending_while_other_stale_rows_still_fail(monkeypatch, tmp_path):
|
||||
ancestor, sibling = os.getpid(), os.getppid()
|
||||
_fleet_homes(monkeypatch, tmp_path, {
|
||||
"default": {"pid": ancestor, "gateway_state": "running", "code_sha": OLD},
|
||||
"ops": {"pid": sibling, "gateway_state": "running", "code_sha": OLD},
|
||||
})
|
||||
|
||||
fleet = ur.collect_fleet_versions(pre_restart_pids=[ancestor, sibling], self_restart_pending={ancestor})
|
||||
states = {row["profile"]: row["state"] for row in fleet}
|
||||
assert states == {"default": ur.RESTART_PENDING_STATE, "ops": "stale"}
|
||||
with contextlib.redirect_stdout(io.StringIO()) as out:
|
||||
assert ur.print_fleet_version_matrix(fleet) is True # the sibling still fails the update
|
||||
assert "restart pending" in out.getvalue() and "1 gateway(s) still running the old code" in out.getvalue()
|
||||
|
||||
# Without the sibling, the pending row alone is not a failure — and the same pid NOT recorded
|
||||
# as pending (the restart phase never reached the ancestor branch) is a plain stale verdict.
|
||||
only_ancestor = [row for row in fleet if row["profile"] == "default"]
|
||||
with contextlib.redirect_stdout(io.StringIO()):
|
||||
assert ur.print_fleet_version_matrix(only_ancestor) is False
|
||||
plain = ur.collect_fleet_versions(pre_restart_pids=[ancestor, sibling])
|
||||
assert {row["state"] for row in plain} == {"stale"}
|
||||
|
||||
|
||||
def test_restart_phase_records_accepted_self_restart_and_verify_exits_clean(monkeypatch, tmp_path):
|
||||
"""The ancestor branch of the drain triage feeds the pid set the matrix reads: end to end the
|
||||
verify phase exits 0 (not 1) and clears the pending marker for an update whose only old-code
|
||||
gateway is the one it runs inside."""
|
||||
import hermes_cli.gateway as gateway
|
||||
import hermes_cli.update_cmd as update_cmd
|
||||
|
||||
ancestor = os.getpid()
|
||||
monkeypatch.setattr(gateway, "_is_pid_ancestor_of_current_process", lambda pid: pid == ancestor)
|
||||
monkeypatch.setattr(gateway, "_request_gateway_self_restart", lambda pid: True)
|
||||
pending: set = set()
|
||||
with contextlib.redirect_stdout(io.StringIO()):
|
||||
assert fleet_mod._drain_or_signal_gateway_for_update(ancestor, 5.0, "default", self_restart_pending=pending)
|
||||
assert pending == {ancestor}
|
||||
|
||||
_fleet_homes(monkeypatch, tmp_path, {"default": {"pid": ancestor, "gateway_state": "running", "code_sha": OLD}})
|
||||
monkeypatch.setattr(fleet_mod, "_print_legacy_units_warning", lambda: None)
|
||||
monkeypatch.setattr(update_cmd, "_finish_dashboard_update_cleanup", lambda *a, **k: None)
|
||||
monkeypatch.setattr(update_cmd, "_surviving_pre_update_serve_runtimes", lambda plan: [])
|
||||
monkeypatch.setattr(fleet_mod._time, "sleep", lambda s: None)
|
||||
cleared = []
|
||||
monkeypatch.setattr(fleet_mod, "_clear_fleet_restart_pending_marker", lambda: cleared.append(True))
|
||||
monkeypatch.setattr("hermes_cli.gateway_migrate.maybe_auto_migrate_after_update", lambda: None)
|
||||
restart = fleet_mod._GatewayRestartOutcome(
|
||||
incomplete=False, phase_errors=[], pre_restart_gateway_pids=[ancestor], restarted_services=["hermes-gateway"],
|
||||
failed_or_stale_units=[], relaunched_profiles=[], externally_supervised_profiles=[], killed_pids=set(),
|
||||
self_restart_pending_pids=pending,
|
||||
)
|
||||
with contextlib.redirect_stdout(io.StringIO()) as out:
|
||||
fleet_mod._verify_fleet_after_update(
|
||||
restart, _pre_update_plan=None, _windows_gateway_resume=None, node_failures=[], update_complete=True,
|
||||
) # a SystemExit(1) here is the #119597 symptom
|
||||
assert "restart pending" in out.getvalue()
|
||||
assert "Update not complete" not in out.getvalue()
|
||||
assert cleared == [True]
|
||||
with contextlib.redirect_stdout(io.StringIO()), pytest.raises(SystemExit) as exc:
|
||||
restart.self_restart_pending_pids = set() # same fleet, identity not threaded → STALE, exit 1
|
||||
fleet_mod._verify_fleet_after_update(
|
||||
restart, _pre_update_plan=None, _windows_gateway_resume=None, node_failures=[], update_complete=True,
|
||||
)
|
||||
assert exc.value.code == 1
|
||||
Reference in New Issue
Block a user