diff --git a/gateway/run.py b/gateway/run.py index e28ec54c31..f5b6050130 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -5380,9 +5380,47 @@ def _claim_host_gateway_role(force: bool = False) -> None: logger.warning("--force: starting a second gateway although %s owns this host.", hr.describe(owner) if owner else "another process") return + if _owner_is_standalone(): + # COMPOSITION with #118236: `host_attach.decide` sent us here with START precisely because + # the owner is another profile's STANDALONE gateway and will never serve us. Refusing now + # exits 75, the supervisor retries in 5s, and the next claim loses the same race — the host + # lock is per OS user and every gateway takes it, so a second profile can NEVER win. An + # unmigrated fleet would spin forever instead of running. Start beside it and point at the + # one command that converges; multiplex-only is enforced against a MULTIPLEXER owner. + logger.warning( + "Another profile's standalone gateway owns this host (%s); starting beside it rather " + "than retrying a race no second profile can win. Fold every profile onto one gateway " + "with: %s", hr.describe(owner) if owner else "owner unknown", _migrate_command()) + return _refuse_second_host_gateway(owner) +def _migrate_command() -> str: + from hermes_cli.gateway_migrate import MIGRATE_COMMAND + + return MIGRATE_COMMAND + + +def _owner_is_standalone() -> bool: + """True when the host owner answers that it does NOT multiplex (an unmigrated fleet). + + Asked only on the lock-losing path, and any failure answers False: an owner we cannot reach + is treated as a multiplexer, which keeps the second-gateway refusal as the default. + """ + try: + from gateway.host_attach import host_gateway, profile_name_for_home, request_serve_profile + + owner = host_gateway() + if owner is None or owner.pid == os.getpid(): + return False + answered = request_serve_profile(profile_name_for_home(get_hermes_home()), owner=owner) + return bool(answered is not None and answered.standalone) + except Exception: + logger.debug("standalone-owner probe failed; keeping the second-gateway refusal", + exc_info=True) + return False + + def _refuse_second_host_gateway(owner) -> None: """Print the named refusal and exit 75 so a supervisor retries instead of parking the unit.""" from gateway import host_rendezvous as hr diff --git a/tests/gateway/test_host_gateway_lock_refusal.py b/tests/gateway/test_host_gateway_lock_refusal.py index b818f1eca6..73bd8052d0 100644 --- a/tests/gateway/test_host_gateway_lock_refusal.py +++ b/tests/gateway/test_host_gateway_lock_refusal.py @@ -83,3 +83,69 @@ def test_force_still_starts_a_second_gateway_and_an_unusable_lock_dir_is_not_a_r hr, "claim_host_lock", lambda role: (hr.HostLockOutcome.COULD_NOT_OPEN, OSError("read-only file system"))) _claim_host_gateway_role() # no SystemExit + + +@pytest.mark.skipif(sys.platform == "win32", reason="flock-based contention setup") +def test_an_unmigrated_standalone_fleet_starts_beside_the_owner_instead_of_spinning( + host_lock_dir, monkeypatch, caplog, +): + """COMPOSITION with #118236 ('a standalone host owner means START, not a parked unit'). + + That change routes a profile whose host owner is another profile's STANDALONE gateway to + START, because no multiplexer serves it. The host-lock refusal then exits 75, the supervisor + retries in 5s, and the next claim loses the same race: the lock is per OS USER and every + gateway takes it, so a second profile can NEVER win it. Composed, the two correct decisions + are an infinite 5s retry loop for every unmigrated fleet with >=2 profiles — including one + installed with --force. The refusal must not fire for a START that exists precisely because + nothing serves this profile. + """ + import logging + + from gateway import host_rendezvous as hr + from gateway.run import _claim_host_gateway_role + + hr.publish_record(hr.ROLE_GATEWAY, profiles=("default",), home=str(host_lock_dir)) + owner = hr.read_record(hr.ROLE_GATEWAY, include_stale=True) + assert owner is not None + # The owner answers the rescan the way a STANDALONE gateway does: "I do not multiplex." + # Stubbed at the wire answer every tree has, so a tree without the carve-out fails on the + # OUTCOME (SystemExit 75) rather than on a missing symbol. + from gateway.host_attach import HostGateway + standalone_owner = HostGateway(pid=owner.pid + 1, home=host_lock_dir, profiles=("default",), + served_known=True, standalone=True) # another process + monkeypatch.setattr("gateway.host_attach.host_gateway", + lambda **kw: standalone_owner) + monkeypatch.setattr("gateway.host_attach.request_serve_profile", + lambda profile, owner=None: standalone_owner) + + handle = _hold_host_lock_from_another_description(hr) + try: + with caplog.at_level(logging.WARNING): + _claim_host_gateway_role() # must NOT SystemExit: 75 here is an unwinnable retry + finally: + handle.close() + + from hermes_cli.gateway_migrate import MIGRATE_COMMAND + logged = "\n".join(r.getMessage() for r in caplog.records) + assert "standalone gateway owns this host" in logged + assert MIGRATE_COMMAND in logged, "the bounded outcome must name the command that converges" + + +@pytest.mark.skipif(sys.platform == "win32", reason="flock-based contention setup") +def test_a_multiplexing_owner_is_still_refused(host_lock_dir, monkeypatch): + """The carve-out is scoped to an unmigrated fleet: losing the race to a MULTIPLEXER is still + the second-gateway shape, and an owner we cannot interrogate is treated as one.""" + from gateway import host_rendezvous as hr + from gateway.restart import GATEWAY_SERVICE_RESTART_EXIT_CODE + from gateway.run import _claim_host_gateway_role + + hr.publish_record(hr.ROLE_GATEWAY, profiles=("default", "coder"), home=str(host_lock_dir)) + monkeypatch.setattr("gateway.host_attach.request_serve_profile", + lambda profile, owner=None: None) # owner never answers + handle = _hold_host_lock_from_another_description(hr) + try: + with pytest.raises(SystemExit) as exc: + _claim_host_gateway_role() + finally: + handle.close() + assert exc.value.code == GATEWAY_SERVICE_RESTART_EXIT_CODE