From d2ef7db751afcbee20120d124c41fe871c54b7b1 Mon Sep 17 00:00:00 2001 From: Austin Pickett Date: Wed, 23 Sep 2026 10:19:39 -0400 Subject: [PATCH] fix(gateway): one served profile's failing chore no longer skips the rest (#120268) _for_each_served_profile ran a housekeeping body once per served profile with no boundary between them, and _housekeeping_chore only catches at the tick level. One profile's unreadable store or broken .env therefore ended the loop, and every profile after it lost its state.db archive/prune, curator, skill-sync and MCP reconcile pass on every tick. That state is reachable: _init_session_db tolerates a failed launch store and keeps running, and hermes serve defers each served profile's auto-archive to this loop (#117746), so a satellite behind a broken launch store had no sweeper at all. Each profile now gets its own try/except, logged at debug like _housekeeping_chore. Co-authored-by: Baris Sencan --- gateway/run_profile_reconcile.py | 11 +++- .../test_housekeeping_profile_scope.py | 65 +++++++++++++++++++ 2 files changed, 74 insertions(+), 2 deletions(-) diff --git a/gateway/run_profile_reconcile.py b/gateway/run_profile_reconcile.py index e0f32bd5e1..473d217dc5 100644 --- a/gateway/run_profile_reconcile.py +++ b/gateway/run_profile_reconcile.py @@ -305,8 +305,15 @@ def _for_each_served_profile(runner, body) -> None: body("default") return for profile_name, profile_home in _multiplex_profile_homes(config): - with _profile_runtime_scope(Path(profile_home)): - body(str(profile_name)) + # One boundary per profile: callers (``_housekeeping_chore``) catch only at the tick level, + # so one profile's unreadable store or broken .env abandoned every profile after it, on + # every tick. The launch store failing is reachable (``_init_session_db`` tolerates it and + # keeps running), and serve defers every served profile's sweep to this loop. + try: + with _profile_runtime_scope(Path(profile_home)): + body(str(profile_name)) + except Exception as exc: + logger.debug("Housekeeping for profile %s skipped: %s", profile_name, exc) def profile_scoped_chore(runner, chore): diff --git a/tests/gateway/test_housekeeping_profile_scope.py b/tests/gateway/test_housekeeping_profile_scope.py index 603218f796..9bdbf68ebf 100644 --- a/tests/gateway/test_housekeeping_profile_scope.py +++ b/tests/gateway/test_housekeeping_profile_scope.py @@ -178,6 +178,71 @@ def test_multiplexed_maintenance_tick_prunes_every_served_profile_store(two_home db.close() +def test_a_failing_profile_does_not_strand_the_profiles_after_it(two_homes, monkeypatch): + """One served profile's broken store must not cost every profile after it its maintenance. + + ``_housekeeping_chore`` catches at the tick level only, so the launch store raising in + ``acquire()`` (reachable: ``_init_session_db`` tolerates a failed primary store and keeps + running) ended the per-profile loop before B, on every tick. Serve defers each served + profile's sweep to this loop, so B had no archiver at all. + """ + import hermes_state_registry as registry + from agent.secret_scope import set_multiplex_active + from hermes_constants import get_hermes_home + from hermes_state import SessionDB + + a, b = two_homes + swept: list = [] + real_acquire = registry.acquire + + def _acquire(*args, **kwargs): + if get_hermes_home() == a: + raise OSError("launch store unavailable") + return real_acquire(*args, **kwargs) + + monkeypatch.setattr(registry, "acquire", _acquire) + monkeypatch.setattr( + SessionDB, "maybe_auto_archive", lambda self, **kw: swept.append(Path(self.db_path))) + monkeypatch.setattr( + "hermes_cli.config.load_config", + lambda *args, **kwargs: {"sessions": {"auto_archive": True, "min_interval_hours": 0}}) + + set_multiplex_active(True) + try: + _run_60_ticks(SimpleNamespace(config=SimpleNamespace(multiplex_profiles=True))) + finally: + set_multiplex_active(False) + + assert swept == [b / "state.db"] + assert get_hermes_home() == a + + +def test_profile_scope_setup_failure_restores_the_callers_home(two_homes, monkeypatch): + """A profile scope whose secret hydration raises must not leave its home installed. + + The home override was set before hydration and only reset in the ``finally`` around the + ``yield``, so a raising ``.env`` load left the housekeeping thread (or a turn's context) + resolving ``get_hermes_home()`` to the failed profile for every later unscoped read. + """ + from agent.secret_scope import current_secret_scope + from hermes_constants import get_hermes_home + + a, b = two_homes + scope_before = current_secret_scope() + + def _boom(home): + raise OSError(f"cannot read {home}/.env") + + monkeypatch.setattr(gateway_run, "_load_profile_secret_scope", _boom) + + with pytest.raises(OSError): + with gateway_run._profile_runtime_scope(b): + pass + + assert get_hermes_home() == a + assert current_secret_scope() == scope_before + + def test_prune_unlinks_transcripts_under_the_configured_sessions_dir(two_homes, tmp_path): """``gateway.sessions_dir`` governs the LAUNCH profile's transcripts; others use their own home.