fix(cli): note pre_restart_pids' per-PID data model gap, pin the matching-start_time path

Addresses the two follow-up notes from review: document that
pre_restart_pids is a bare PID set (not (pid, start_time) pairs), so a
recycled PID from one gateway landing in another's stale record could
still mislabel it as down; and add a companion test asserting a
matching start_time still yields the live/current row.
This commit is contained in:
chelsealong
2026-08-24 16:18:13 +00:00
committed by Teknium
parent 5d7ed70eef
commit 2812d6121b
2 changed files with 27 additions and 0 deletions

View File

@@ -391,6 +391,12 @@ def collect_fleet_versions(
# stop, startup failure, stale record from a long-dead
# gateway) keeps the historical no-row behavior so the
# feature's rollout can't false-positive.
#
# ``_pre_restart`` is a bare set of PIDs, not (pid, start_time)
# pairs, so a recycled PID from gateway A landing in B's stale
# record could still mislabel B as down if A's PID happened to
# be in the pre-restart snapshot — inherent to the snapshot's
# data model, not something this guard can fix on its own.
gw_state = record.get("gateway_state")
if (
pid in _pre_restart

View File

@@ -95,6 +95,27 @@ def test_recycled_pid_is_not_reported_stale(monkeypatch, tmp_path):
assert fleet[0]["state"] == "down"
def test_matching_start_time_is_still_live(monkeypatch, tmp_path):
"""A record whose start_time matches the live process is not recycled."""
from gateway.status import _get_process_start_time
pid = os.getpid()
_setup(
monkeypatch,
tmp_path,
{
"pid": pid,
"start_time": _get_process_start_time(pid),
"gateway_state": "running",
"code_sha": "HEADSHA",
"kind": "hermes-gateway",
},
)
fleet = ur.collect_fleet_versions(pre_restart_pids=[pid])
assert len(fleet) == 1
assert fleet[0]["state"] == "current"
def test_live_gateway_rows_unchanged(monkeypatch, tmp_path):
"""The live-pid path is untouched by the down-state addition."""
_setup(