fix: count service-supervised gateways as running for per-profile cold-start

The per-profile cold-start probe took its running set from the socket-paused
ordinary gateways only. A profile whose gateway is alive under an SCM service
is skipped by the socket pause, so it looked "not running"; with an empty
current-PID list its live start attestation read as dead and the profile
landed in cold_start_profiles. On resume the service was restarted AND a
second, unsupervised gateway was spawned for the same profile.

Build the running set from every profile that had ANY live gateway at
discovery time: paused profiles, profile-mapped processes, and service
gateways' profiles.

Review finding: service-supervised running profile was cold-started beside
its restarted SCM service (double gateway).
This commit is contained in:
teknium1
2026-09-14 21:05:58 -07:00
committed by Teknium
parent b19e5786e2
commit 27df47af5e
2 changed files with 27 additions and 1 deletions

View File

@@ -923,7 +923,10 @@ def _pause_windows_gateways_for_update() -> dict | None:
if any(not u.get("argv") for u in unmapped): # no recoverable cmdline (psutil missing, denied, gone)
print(" Restart manually after update: hermes gateway run")
token = {"resume_needed": True, "profiles": profiles, "unmapped_pids": unmapped_pids, "unmapped": unmapped}
_record_attested_cold_start_profiles(token, set(profiles))
# Every profile with ANY live gateway at discovery counts as running: service-supervised ones skip the
# socket pause (absent from ``profiles``) but the SCM restart brings them back, not a cold-start.
running_profiles = set(profiles) | {str(p.profile) for p in profile_processes.values()} | {str(s.profile) for s in service_gateways}
_record_attested_cold_start_profiles(token, running_profiles)
return _pause_windows_gateway_services(service_gateways, token, profiles, unmapped)

View File

@@ -322,3 +322,26 @@ def test_every_dead_attested_profile_is_cold_started_when_nothing_runs(monkeypat
assert order == ["active", homes["beta"]]
assert json.loads(beta_marker.read_text(encoding="utf-8"))["generation"] != beta_generation
assert token["resume_needed"] is False
def test_service_supervised_running_profile_is_not_cold_started(monkeypatch, tmp_path):
"""A profile whose gateway is alive under an SCM service is skipped by the socket pause, so it is
absent from ``token["profiles"]``; it must still count as RUNNING for the per-profile probe or its
live attestation reads as dead and resume spawns a second, unsupervised gateway beside the
restarted service."""
from types import SimpleNamespace
homes = _running_beta_pause_fixture(monkeypatch, tmp_path)
svc_proc = SimpleNamespace(pid=900, profile="beta", path=homes["beta"])
service = SimpleNamespace(name="HermesGw-beta", profile="beta", service_pid=800, gateway_pid=900,
descendant_identities=(), service_create_time=1.0, gateway_create_time=2.0)
monkeypatch.setattr(update_cmd_windows, "_discover_windows_gateways", lambda: ({900: svc_proc}, [service], {900}, [900]))
monkeypatch.setattr(update_cmd_windows, "_request_socket_pauses", lambda *a: ({}, [], []))
monkeypatch.setattr(update_cmd, "_stop_windows_gateway_service", lambda *a, **k: None)
gateway_windows._write_start_attestation([900], "direct spawn (PID 900)", home=homes["beta"])
token = update_cmd._pause_windows_gateways_for_update()
assert token["services"] == ["HermesGw-beta"]
assert token["profiles"] == {}
assert "cold_start_profiles" not in token