Files
hermes-agent/tests/gateway/test_heartbeat_watch_restore.py
carlotestor 3e88c8032d perf(gateway): enter each profile scope once per heartbeat-restore scan; signature-cache profile config.yaml parses
restore_heartbeat_watches entered _profile_scope_for_source for every routed
session on every poll. Each entry hydrated the profile secret scope and rebuilt
the terminal policy, and both re-parsed the profile config.yaml from disk, so N
routed sessions cost 2N YAML parses per poll even though nothing changed.

- Group entries by resolved profile home and enter the scope once per group.
- Add utils.load_yaml_file_readonly (file_signature-keyed cache) and use it in
  env_loader._load_secrets_config and terminal_scope.build_profile_terminal_scope,
  which were both open()+fast_safe_load per scope entry. Present-but-unparseable
  still fails closed: parse errors propagate and are never cached.

Measured on a 3-profile host: one _profile_runtime_scope enter/exit 1.80 ms -> 0.11 ms.

(cherry picked from commit c6b16629bd38799bbf166206c73a6140a9559a61)
2026-09-22 16:15:34 +05:30

229 lines
11 KiB
Python

"""Restart recovery uses current routing and never borrows another profile's heartbeat."""
import asyncio
from pathlib import Path
from types import SimpleNamespace
from unittest.mock import AsyncMock
import pytest
from gateway.config import GatewayConfig, Platform
from gateway.run import GatewayRunner, _profile_runtime_scope
from gateway.session import SessionStore, SessionSource
from hermes_cli import goals
from hermes_cli.heartbeat import HeartbeatManager, HeartbeatState
from hermes_state import SessionDB
@pytest.mark.asyncio
async def test_restore_retries_persisted_routes_in_their_own_profiles(tmp_path, monkeypatch):
from gateway.run_heartbeat_restore import restore_heartbeat_watches
home = tmp_path / '.hermes'
named = home / 'profiles' / 'work'
named.mkdir(parents=True)
(named / 'config.yaml').write_text('{}')
monkeypatch.setattr(Path, 'home', lambda: tmp_path)
monkeypatch.setenv('HERMES_HOME', str(home))
dbs = {str(p): SessionDB(db_path=p / 'state.db') for p in (home, named)}
monkeypatch.setattr(goals, '_DB_CACHE', dbs)
config = GatewayConfig(multiplex_profiles=True)
store = SessionStore(home / 'sessions', config)
entries = []
try:
for profile, status, topic in [(None, 'active', '11'), ('work', 'active', '22'),
('work', 'paused', '33'), (None, 'cleared', '44')]:
source = SessionSource(platform=Platform.TELEGRAM, chat_id='chat',
thread_id=topic, profile=profile, scope_id='workspace')
with _profile_runtime_scope(named if profile else home):
entry = store.get_or_create_session(source)
manager = HeartbeatManager(entry.session_id)
manager.set('check ' + topic, 60)
if status == 'paused':
manager.pause()
elif status == 'cleared':
manager.clear()
entries.append(entry)
# Reload the real persisted routing index as a fresh process would.
store.close_all_db_handles()
runner = GatewayRunner.__new__(GatewayRunner)
runner.config = config
runner.session_store = SessionStore(home / 'sessions', config)
runner._heartbeat_watch = {}
runner._start_heartbeat_poller = lambda: None
runner._profile_name_for_source = lambda source, adapter_profile=None: source.profile
runner._delivery_adapter_for = lambda source: object()
runner._run_in_executor_with_context = asyncio.to_thread
original = runner.session_store.list_sessions
with monkeypatch.context() as patch:
patch.setattr(runner.session_store, 'list_sessions', lambda: (_ for _ in ()).throw(OSError('cold index')))
await restore_heartbeat_watches(runner)
assert runner._heartbeat_watch == {}
assert original()
# A poller inherited from work must still restore the default profile too.
with _profile_runtime_scope(named):
await restore_heartbeat_watches(runner)
expected = {e.session_key: (e.origin, e.session_id) for e in entries[:2]}
assert runner._heartbeat_watch == expected
with monkeypatch.context() as patch:
patch.setattr('hermes_cli.heartbeat.load_heartbeat', lambda sid: None)
await restore_heartbeat_watches(runner)
assert runner._heartbeat_watch == expected
runner._heartbeat_watch.clear()
await restore_heartbeat_watches(runner)
assert runner._heartbeat_watch == expected
finally:
store.close_all_db_handles()
if 'runner' in locals():
runner.session_store.close_all_db_handles()
for db in dbs.values():
db.close()
@pytest.mark.asyncio
async def test_restore_skips_session_sweep_when_no_heartbeats_exist(tmp_path, monkeypatch):
"""Idle case: no ACTIVE ``heartbeat:*`` row in any served profile → the restore poll must not
sweep the routing index (each swept origin re-parses that profile's config/secrets). An active
row, or a store the probe cannot read, brings the sweep back."""
from gateway.run_heartbeat_restore import restore_heartbeat_watches
home = tmp_path / '.hermes'
named = home / 'profiles' / 'work'
named.mkdir(parents=True)
(named / 'config.yaml').write_text('{}')
monkeypatch.setattr(Path, 'home', lambda: tmp_path)
monkeypatch.setenv('HERMES_HOME', str(home))
dbs = {str(p): SessionDB(db_path=p / 'state.db') for p in (home, named)}
monkeypatch.setattr(goals, '_DB_CACHE', dbs)
config = GatewayConfig(multiplex_profiles=True)
store = SessionStore(home / 'sessions', config)
try:
source = SessionSource(platform=Platform.TELEGRAM, chat_id='chat',
thread_id='7', profile=None, scope_id=None)
store.get_or_create_session(source)
runner = GatewayRunner.__new__(GatewayRunner)
runner.config = config
runner.session_store = store
runner._heartbeat_watch = {}
runner._run_in_executor_with_context = asyncio.to_thread
sweeps = []
original = store.list_sessions
monkeypatch.setattr(store, 'list_sessions',
lambda: (sweeps.append(1), original())[1])
await restore_heartbeat_watches(runner)
assert sweeps == []
assert runner._heartbeat_watch == {}
# A cleared heartbeat keeps its row (status=cleared): still nothing to restore.
dbs[str(named)].set_meta('heartbeat:old', HeartbeatState(
prompt='p', interval_seconds=60, status='cleared').to_json())
await restore_heartbeat_watches(runner)
assert sweeps == []
# An ACTIVE row in the secondary store opens the gate.
dbs[str(named)].set_meta('heartbeat:live', HeartbeatState(
prompt='p', interval_seconds=60, status='active').to_json())
await restore_heartbeat_watches(runner)
assert sweeps == [1]
# A `-p work` multiplexer's routing home is the named profile; an active heartbeat that
# lives only in the DEFAULT store must still bring the sweep back.
dbs[str(named)].set_meta('heartbeat:live', HeartbeatState(
prompt='p', interval_seconds=60, status='cleared').to_json())
store._routing_home = named
await restore_heartbeat_watches(runner)
assert sweeps == [1], 'no active row anywhere: still idle'
dbs[str(home)].set_meta('heartbeat:root', HeartbeatState(
prompt='p', interval_seconds=60, status='active').to_json())
await restore_heartbeat_watches(runner)
assert sweeps == [1, 1], 'active row in the default store must be seen from a named routing home'
dbs[str(home)].set_meta('heartbeat:root', HeartbeatState(
prompt='p', interval_seconds=60, status='cleared').to_json())
# Fail OPEN: a store the probe cannot open must not suppress the sweep.
monkeypatch.setattr('gateway.run_idle_gates._profile_session_db_probe', lambda _home: None)
await restore_heartbeat_watches(runner)
assert sweeps == [1, 1, 1]
finally:
store.close_all_db_handles()
for db in dbs.values():
db.close()
@pytest.mark.asyncio
async def test_startup_arms_retry_poller_even_without_any_watches(monkeypatch):
runner = GatewayRunner.__new__(GatewayRunner)
runner._heartbeat_watch = {}
runner._background_tasks = set()
runner._ensure_hosted_room_worker = AsyncMock()
runner._hosted_room_worker_watcher = AsyncMock()
runner._spawn_supervised = lambda *args: None
runner._start_loop_heartbeat_task = lambda: None
runner.hooks = SimpleNamespace(loaded_hooks=[], emit=AsyncMock())
runner.adapters = {}
runner._send_update_notification = AsyncMock(return_value=True)
runner.session_store = SimpleNamespace(list_sessions=lambda: [])
runner._run_in_executor_with_context = asyncio.to_thread
monkeypatch.setattr('gateway.channel_directory.build_channel_directory', AsyncMock(return_value={}))
await runner._start_post_connect_services(0)
task = getattr(runner, '_heartbeat_poll_task', None)
try:
assert task is not None and not task.done()
assert runner._heartbeat_watch == {}
finally:
if task:
task.cancel()
await asyncio.gather(task, return_exceptions=True)
@pytest.mark.asyncio
async def test_restore_enters_each_profile_scope_once_per_scan(tmp_path, monkeypatch):
"""N routed sessions in one profile cost one scope entry, not N (scope entry re-parses config)."""
from gateway import run_heartbeat_restore
from gateway.run_heartbeat_restore import restore_heartbeat_watches
home = tmp_path / '.hermes'
named = home / 'profiles' / 'work'
named.mkdir(parents=True)
(named / 'config.yaml').write_text('{}')
monkeypatch.setattr(Path, 'home', lambda: tmp_path)
monkeypatch.setenv('HERMES_HOME', str(home))
dbs = {str(p): SessionDB(db_path=p / 'state.db') for p in (home, named)}
monkeypatch.setattr(goals, '_DB_CACHE', dbs)
config = GatewayConfig(multiplex_profiles=True)
store = SessionStore(home / 'sessions', config)
try:
expected = {}
for profile in (None, 'work', 'work', None, 'work'):
source = SessionSource(platform=Platform.TELEGRAM, chat_id='chat',
thread_id=str(len(expected)), profile=profile, scope_id='workspace')
with _profile_runtime_scope(named if profile else home):
entry = store.get_or_create_session(source)
HeartbeatManager(entry.session_id).set('check', 60)
expected[entry.session_key] = (entry.origin, entry.session_id)
store.close_all_db_handles()
runner = GatewayRunner.__new__(GatewayRunner)
runner.config = config
runner.session_store = SessionStore(home / 'sessions', config)
runner._heartbeat_watch = {}
runner._start_heartbeat_poller = lambda: None
runner._profile_name_for_source = lambda source: source.profile
runner._adapter_for_source = lambda source: object()
runner._run_in_executor_with_context = asyncio.to_thread
entered = []
real = GatewayRunner._profile_scope_for_source
def counting(self, source):
entered.append(source.profile)
return real(self, source)
monkeypatch.setattr(GatewayRunner, '_profile_scope_for_source', counting)
await restore_heartbeat_watches(runner)
assert runner._heartbeat_watch == expected
assert sorted(entered, key=str) == [None, 'work']
finally:
store.close_all_db_handles()
if 'runner' in locals():
runner.session_store.close_all_db_handles()
for db in dbs.values():
db.close()