From 060cd7f9bb5041ca77eb3374829becffb332f2ab Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 9 Sep 2026 05:31:07 -0700 Subject: [PATCH] fix(state): trim the futile-holder FTS diagnostic to shape and fix the remedy text MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Slim redo of the mechanism from #106410 on top of its pick (no wrappers, no persisted "kind" enum, no process-local flag that dies with the process): - Futility = the SAME holder PID set has blocked >= _FTS_HOLDER_FUTILE_ATTEMPTS (10) deferrals over >= _FTS_HOLDER_FUTILE_SECONDS (30 min); tracked as holders_since/holders_attempts in the persisted fts_rebuild_deferral record and reset whenever the holder set changes. The 3-deferral/60 s escalate window is the orphan-reap gate and stays as is. - ONE escalated ERROR line names each holder pid + cmdline and the remedy that can actually be followed from inside a gateway session: stop ONLY the other holder; this process's own retry admits the rebuild within 60 s. The old "with the gateway stopped" advice was unrunnable from a gateway-hosted session (the gateway is the session) and is gone from both log and doctor. - hermes doctor renders the futile record distinctly. - retry_deferred_fts_recovery: a capped backoff earned by holder set X no longer applies once the live holder set differs from X, so stopping the other service is followed by a retry on the next tick, not up to an hour later (the issue's 16-min wait). - Tests trimmed from 5 to 2 invariants (futile line + doctor entry after N same-holder deferrals; backoff reset when the holder set changes); the contributor's control tests for changing PIDs / orphan reap are covered by the existing test_repeated_deferrals_reap_inactive_orphan_then_rebuild. The "canonical writes and LIKE search remain available" WARNING is kept because it is true on origin/main: a stale open drops every FTS trigger, so the messages INSERT succeeds (probed live with a real state.db + a second process holding it). Writes fail only when a peer re-publishes triggers over the corrupt index — a separate class, not this diagnostic. Refs #106393 --- hermes_cli/doctor_state.py | 24 +- hermes_state_schema.py | 115 ++++----- tests/state/test_fts_runtime_rebuild.py | 306 ++++++------------------ tests/test_state_db_stats.py | 26 -- 4 files changed, 139 insertions(+), 332 deletions(-) diff --git a/hermes_cli/doctor_state.py b/hermes_cli/doctor_state.py index aba3d59b5a..8851ee4b7c 100644 --- a/hermes_cli/doctor_state.py +++ b/hermes_cli/doctor_state.py @@ -80,21 +80,17 @@ def _render_state_db_stats(stats: dict, holders=None) -> list: lines.append(("info", "FTS tables: " + (", ".join(present) if present else "none"), "")) deferral = stats.get("fts_rebuild_deferral") if isinstance(deferral, dict): - pids = deferral.get("holder_pids") or [] - attempts = deferral.get("attempts") or "?" - futile = deferral.get("futile") is True or deferral.get("kind") == "permanent_holder" - if futile: - lines.append(( - "warn", - f"state.db FTS repair is blocked by a permanent holder " - f"(another Hermes service, PID(s) {pids}) after {attempts} futile deferral(s)", - "(stop the other Hermes service; leave this gateway running — " - "retry_deferred_fts_recovery admits once this process is the sole holder)", - )) + pids = deferral.get("holder_pids") or "unknown" + if deferral.get("futile"): + lines.append(("warn", f"state.db FTS repair is blocked by the same holder(s) PID(s) {pids} for " + f"{deferral.get('holders_attempts') or '?'} consecutive deferral(s); waiting is futile", + "(stop ONLY the listed process(es) — the gateway keeps running and its own retry " + "rebuilds within a minute of the holder leaving)")) else: - lines.append(("warn", f"state.db FTS repair is blocked after {attempts} deferral(s) " - f"by PID(s) {pids or 'unknown'}", - "(stop the listed processes, then run 'hermes sessions optimize-storage' with the gateway stopped)")) + lines.append(("warn", f"state.db FTS repair is blocked after {deferral.get('attempts') or '?'} deferral(s) " + f"by PID(s) {pids}", + "(stop the listed processes; the gateway's own retry then rebuilds, or run " + "'hermes sessions optimize-storage' with every holder stopped)")) # Oversized DB: suggest auto_prune, plus the offline optimize-storage pass when the FTS rebuild is # pending OR the DB predates the current trigram layout (fts_storage_version < FTS_STORAGE_VERSION). if logical is not None and logical > STATE_DB_SIZE_WARN_BYTES: diff --git a/hermes_state_schema.py b/hermes_state_schema.py index 87414606e5..65a4526bac 100644 --- a/hermes_state_schema.py +++ b/hermes_state_schema.py @@ -26,31 +26,26 @@ from hermes_state_common import ( LEGACY_FTS_TRIGRAM_SQL, SCHEMA_SQL, SCHEMA_VERSION, _FTS_CJK_TRIGGERS, _FTS_TRIGGERS, _ephemeral_child_sql, fts_rebuild_admission, ) +from hermes_state_holders import _read_proc_argv # Pre-split logger identity so log filtering/capture is unchanged. logger = logging.getLogger("hermes_state") _FTS_HOLDER_ESCALATE_ATTEMPTS = 3 _FTS_HOLDER_ESCALATE_SECONDS = 60.0 +# The same holder PID set blocking this many deferrals over this long is a structurally resident +# peer (a supervised service on the same HERMES_HOME), not a transient one worth waiting out (#106393). +_FTS_HOLDER_FUTILE_ATTEMPTS = 10 +_FTS_HOLDER_FUTILE_SECONDS = 1800.0 # retry_deferred_fts_recovery cadence: startup paid the full admission wait once; later # retries are non-blocking probes whose spacing doubles up to the cap. _FTS_STALE_RETRY_SECONDS = 60.0 _FTS_STALE_RETRY_MAX_SECONDS = 3600.0 -def _fts_holder_pid_set(value): - """Sorted unique positive PIDs, or None when the stored set is missing/unusable.""" - if not isinstance(value, (list, tuple)): - return None - pids = set() - for item in value: - try: - pid = int(item) - except (TypeError, ValueError): - return None - if pid > 0: - pids.add(pid) - return sorted(pids) +def _holder_cmdline(pid: int) -> str: + argv = _read_proc_argv(pid) + return " ".join(argv)[:120] if argv else "" # schema_read_probe_statements() cache (parses SCHEMA_SQL in an in-memory DB; once per process). _READ_PROBE_STATEMENTS: Optional[tuple] = None @@ -437,8 +432,12 @@ class SessionSchemaMixin: """Record a deferral diagnostic for the foreign processes holding the DB; True = defer (holders remain). After ``_FTS_HOLDER_ESCALATE_ATTEMPTS`` deferrals spanning ``_FTS_HOLDER_ESCALATE_SECONDS``, provably inactive orphan Desktop backends are - reaped and the holders re-checked. A stable PID set that survives that reap is - recorded as a futile/permanent-holder diagnostic; the orphan predicate is unchanged.""" + reaped and the holders re-checked. The orphan reap is the only exit, so a supervised + peer (never an orphan) blocks forever: once the SAME PID set has blocked + ``_FTS_HOLDER_FUTILE_ATTEMPTS`` deferrals over ``_FTS_HOLDER_FUTILE_SECONDS`` the + record is marked ``futile`` and the escalation names the holders and the remedy that + works from inside a gateway session (stop only the other holder; this process's own + retry tick admits the rebuild). A changed holder set restarts that window.""" now = time.time() try: row = cursor.execute( @@ -451,13 +450,15 @@ class SessionSchemaMixin: try: first_seen = float(record.get("first_seen", now)) attempts = int(record.get("attempts", 0)) + 1 + holders_since = float(record.get("holders_since", now)) + holders_attempts = int(record.get("holders_attempts", 0)) + 1 except (TypeError, ValueError): - first_seen, attempts = now, 1 + first_seen, attempts, holders_since, holders_attempts = now, 1, now, 1 if first_seen > now or first_seen < 0: first_seen = now - previous_pids = _fts_holder_pid_set(record.get("holder_pids")) holder_pids = sorted({pid for pid, _path in foreign_holders if pid > 0}) - futile = False + if holder_pids != record.get("holder_pids"): + holders_since, holders_attempts = now, 1 if attempts >= _FTS_HOLDER_ESCALATE_ATTEMPTS and now - first_seen >= _FTS_HOLDER_ESCALATE_SECONDS: reaped = self._reap_inactive_orphan_desktop_holders( foreign_holders, min_age_seconds=_FTS_HOLDER_ESCALATE_SECONDS, @@ -469,58 +470,40 @@ class SessionSchemaMixin: ) foreign_holders = self._foreign_state_db_holders() holder_pids = sorted({pid for pid, _path in foreign_holders if pid > 0}) - if foreign_holders: - if previous_pids is not None and holder_pids and holder_pids == previous_pids: - futile = True - logger.error( - "state.db FTS repair is futile after %d deferrals: the same " - "holder PID set %s is a permanent holder (another Hermes " - "service), not a transient peer. Stop the other Hermes " - "service; leave this gateway running. " - "retry_deferred_fts_recovery admits once this process is " - "the sole holder. `hermes doctor` reports this degraded state.", - attempts, holder_pids, - ) - else: - logger.error( - "state.db FTS repair remains blocked after %d deferrals " - "by holder(s) %s. Stop the listed processes, then run " - "`hermes sessions optimize-storage` with the gateway stopped. " - "`hermes doctor` reports this degraded state.", attempts, foreign_holders, - ) - else: - self._fts_stale_retry_after = 0.0 - self._fts_stale_retry_interval = 0.0 - self._fts_permanent_holder_deferral = False + futile = bool(holder_pids) and ( + holders_attempts >= _FTS_HOLDER_FUTILE_ATTEMPTS and now - holders_since >= _FTS_HOLDER_FUTILE_SECONDS + ) diagnostic = { - "first_seen": first_seen, "last_seen": now, "attempts": attempts, - "holder_pids": holder_pids, + "first_seen": first_seen, "last_seen": now, "attempts": attempts, "holder_pids": holder_pids, + "holders_since": holders_since, "holders_attempts": holders_attempts, "futile": futile, } - if futile: - diagnostic["futile"] = True - diagnostic["kind"] = "permanent_holder" - self._fts_permanent_holder_deferral = True - else: - self._fts_permanent_holder_deferral = False cursor.execute( "INSERT INTO state_meta (key, value) VALUES (?, ?) ON CONFLICT(key) DO UPDATE SET value = excluded.value", (FTS_REBUILD_DEFERRAL_KEY, json.dumps(diagnostic, sort_keys=True)), ) if not foreign_holders: + self._fts_deferred_holder_pids = None return False + self._fts_deferred_holder_pids = holder_pids if futile: - logger.warning( - "Deferred stale state.db FTS rebuild while a permanent foreign holder " - "blocks repair (%s); stop the other Hermes service and leave this " - "gateway running (deferral %d).", - foreign_holders, attempts, + logger.error( + "state.db FTS repair has been blocked by the same holder(s) for %d deferrals over %.0f min " + "(%s); waiting is futile. Stop ONLY the other holder(s) — this process keeps running and its " + "own retry admits the rebuild within %.0fs of the holder leaving. `hermes doctor` shows this.", + holders_attempts, (now - holders_since) / 60.0, + ", ".join(f"pid {pid}: {_holder_cmdline(pid)}" for pid in holder_pids), _FTS_STALE_RETRY_SECONDS, ) - else: - logger.warning( - "Deferred stale state.db FTS rebuild while foreign processes " - "hold the database or WAL sidecars (%s); canonical writes and LIKE search remain available (deferral %d).", - foreign_holders, attempts, + elif attempts >= _FTS_HOLDER_ESCALATE_ATTEMPTS and now - first_seen >= _FTS_HOLDER_ESCALATE_SECONDS: + logger.error( + "state.db FTS repair remains blocked after %d deferrals by holder(s) %s. Stop the listed " + "processes (this process's own retry then rebuilds), or run `hermes sessions optimize-storage` " + "with every holder stopped. `hermes doctor` reports this degraded state.", attempts, foreign_holders, ) + logger.warning( + "Deferred stale state.db FTS rebuild while foreign processes " + "hold the database or WAL sidecars (%s); canonical writes and LIKE search remain available (deferral %d).", + foreign_holders, attempts, + ) return True def _recover_stale_fts(self, cursor: sqlite3.Cursor, *, legacy: bool, timeout_seconds=None) -> bool: @@ -563,13 +546,15 @@ class SessionSchemaMixin: if self.read_only or self._conn is None: return False now = time.monotonic() - if getattr(self, "_fts_permanent_holder_deferral", False) and not self._foreign_state_db_holders(): - # Permanent holder gone: do not wait out a doubled-to-cap backoff (#106393). - self._fts_stale_retry_after = 0.0 - self._fts_stale_retry_interval = 0.0 - self._fts_permanent_holder_deferral = False + deferred_pids = getattr(self, "_fts_deferred_holder_pids", None) if now < getattr(self, "_fts_stale_retry_after", 0.0): - return False + # The backoff was earned by a specific holder set; once that set changes (the other + # service stopped) a capped backoff would idle up to an hour with nothing blocking (#106393). + if deferred_pids is None or sorted( + {pid for pid, _path in self._foreign_state_db_holders() if pid > 0} + ) == deferred_pids: + return False + self._fts_stale_retry_interval = 0.0 interval = float(getattr(self, "_fts_stale_retry_interval", 0.0)) if interval <= 0.0: interval = _FTS_STALE_RETRY_SECONDS diff --git a/tests/state/test_fts_runtime_rebuild.py b/tests/state/test_fts_runtime_rebuild.py index 38108f59a7..51e4f50cd7 100644 --- a/tests/state/test_fts_runtime_rebuild.py +++ b/tests/state/test_fts_runtime_rebuild.py @@ -14,6 +14,7 @@ rebuild later, outside the failed live write/search operation. """ import json +import logging import os import sqlite3 import time @@ -28,24 +29,6 @@ from hermes_state_common import FTS_REBUILD_DEFERRAL_KEY, FTS_STALE_KEY, LEGACY_ from hermes_state_dbfile import _concrete_state_db_holder_pids, _is_inactive_orphan_desktop_holder -def _deferral_is_futile(record): - if not isinstance(record, dict): - return False - if record.get("futile") is True: - return True - kind = str(record.get("kind") or record.get("status") or "").lower() - return "futile" in kind or "permanent_holder" in kind - - -def _doctor_deferral_blob(stats): - from hermes_cli.doctor_state import _render_state_db_stats - - return " ".join( - " ".join(str(part) for part in row) - for row in _render_state_db_stats(stats) - ).lower() - - @pytest.fixture def db(tmp_path): d = SessionDB(db_path=tmp_path / "state.db") @@ -804,10 +787,13 @@ class TestRuntimeFtsRebuild: finally: reopened.close() - def test_stable_permanent_holder_marks_futile_deferral( + def test_same_holder_set_across_futile_window_names_holder_and_gateway_safe_remedy( self, db, tmp_path, monkeypatch, caplog ): - """Same supervised holder across the escalate window is futile, not a transient peer.""" + """#106393: a supervised peer never satisfies the orphan reap, so the generic + 'remains blocked ... with the gateway stopped' escalation repeats forever. Once the SAME + PID set has blocked the futile window, the record is marked futile, the escalation names + the holder's cmdline and a remedy runnable from inside the gateway, and doctor says so.""" if not db._fts_enabled: pytest.skip("FTS5 unavailable in this build") db_path = tmp_path / "state.db" @@ -815,229 +801,95 @@ class TestRuntimeFtsRebuild: db.append_message("s1", "user", "seed") _corrupt_fts(db_path) monkeypatch.setattr( - db, - "rebuild_fts", - lambda: (_ for _ in ()).throw(sqlite3.DatabaseError("still corrupt")), + db, "rebuild_fts", lambda: (_ for _ in ()).throw(sqlite3.DatabaseError("still corrupt")), ) db.append_message("s1", "user", "before restart") db.close() - raw = sqlite3.connect(str(db_path)) - raw.execute( - "INSERT INTO state_meta (key, value) VALUES (?, ?) " - "ON CONFLICT(key) DO UPDATE SET value = excluded.value", - ( - FTS_REBUILD_DEFERRAL_KEY, - json.dumps({ - "first_seen": 1.0, - "last_seen": 30.0, - "attempts": 2, - "holder_pids": [4242], - }), - ), - ) - raw.commit() - raw.close() - monkeypatch.setattr( - SessionDB, - "_foreign_state_db_holders", - lambda self: [(4242, str(db_path) + "-wal")], + SessionDB, "_foreign_state_db_holders", lambda self: [(4242, str(db_path) + "-wal")], ) monkeypatch.setattr( - SessionDB, - "_reap_inactive_orphan_desktop_holders", - lambda self, holders, *, min_age_seconds: [], + SessionDB, "_reap_inactive_orphan_desktop_holders", lambda self, holders, *, min_age_seconds: [], ) - monkeypatch.setattr(hermes_state_schema.time, "time", lambda: 120.0) + monkeypatch.setattr( + hermes_state_schema, "_read_proc_argv", lambda pid: ["python", "-m", "hermes_cli.main", "serve"], raising=False, + ) + clock = [1000.0] + monkeypatch.setattr(hermes_state_schema.time, "time", lambda: clock[0]) + from hermes_cli.doctor_state import _render_state_db_stats + from hermes_state_dbfile import collect_state_db_stats + + def doctor_blob(): + return " ".join(" ".join(row) for row in _render_state_db_stats(collect_state_db_stats(db_path))).lower() + + # Read via getattr so the red-on-base run reaches the behavioural assertion, not a NameError. + futile_attempts = getattr(hermes_state_schema, "_FTS_HOLDER_FUTILE_ATTEMPTS", 10) + futile_seconds = getattr(hermes_state_schema, "_FTS_HOLDER_FUTILE_SECONDS", 1800.0) reopened = SessionDB(db_path=db_path) try: + cursor = reopened._conn.cursor() + # One deferral short of the futile window: still the generic escalation. + for _ in range(futile_attempts - 2): + clock[0] += futile_seconds + caplog.clear() + assert reopened._recover_stale_fts(cursor, legacy=False, timeout_seconds=0.0) is False + reopened._conn.commit() + assert not json.loads(_meta_value(db_path, FTS_REBUILD_DEFERRAL_KEY)).get("futile") + assert "waiting is futile" not in caplog.text + assert "futile" not in doctor_blob() + + clock[0] += futile_seconds + caplog.clear() + assert reopened._recover_stale_fts(cursor, legacy=False, timeout_seconds=0.0) is False + reopened._conn.commit() record = json.loads(_meta_value(db_path, FTS_REBUILD_DEFERRAL_KEY)) - assert _deferral_is_futile(record), record - assert record.get("holder_pids") == [4242] + assert record.get("futile") is True and record["holder_pids"] == [4242] + futile_lines = [r for r in caplog.records if "waiting is futile" in r.getMessage()] + assert len(futile_lines) == 1 and futile_lines[0].levelno == logging.ERROR + msg = futile_lines[0].getMessage() + assert "pid 4242: python -m hermes_cli.main serve" in msg + assert "Stop ONLY the other holder" in msg and "with the gateway stopped" not in msg + blob = doctor_blob() + assert "4242" in blob and "waiting is futile" in blob and "stop only" in blob + assert "gateway stopped" not in blob + finally: + reopened.close() + + def test_retry_backoff_resets_when_the_blocking_holder_set_changes( + self, db, tmp_path, monkeypatch + ): + """#106393: days of deferrals pin the retry interval at the 1h cap, so stopping the other + holder was followed by up to an hour of nothing. A changed holder set must retry now.""" + if not db._fts_enabled: + pytest.skip("FTS5 unavailable in this build") + db_path = tmp_path / "state.db" + db.create_session("s1", source="test") + db.append_message("s1", "user", "seed") + _corrupt_fts(db_path) + monkeypatch.setattr( + db, "rebuild_fts", lambda: (_ for _ in ()).throw(sqlite3.DatabaseError("still corrupt")), + ) + db.append_message("s1", "user", "before restart") + db.close() + + holders = [(4242, str(db_path) + "-wal")] + monkeypatch.setattr(SessionDB, "_foreign_state_db_holders", lambda self: list(holders)) + reopened = SessionDB(db_path=db_path) + try: assert reopened._fts_stale is True - from hermes_cli.doctor_state import _render_state_db_stats - from hermes_state_dbfile import collect_state_db_stats - - blob = _doctor_deferral_blob(collect_state_db_stats(db_path)) - assert "stop the other hermes service" in blob - assert "gateway" in blob - assert "canonical writes and like search remain available" not in caplog.text.lower() - finally: - reopened.close() - - def test_changing_holder_pids_do_not_mark_futile( - self, db, tmp_path, monkeypatch - ): - if not db._fts_enabled: - pytest.skip("FTS5 unavailable in this build") - db_path = tmp_path / "state.db" - db.create_session("s1", source="test") - db.append_message("s1", "user", "seed") - _corrupt_fts(db_path) - monkeypatch.setattr( - db, - "rebuild_fts", - lambda: (_ for _ in ()).throw(sqlite3.DatabaseError("still corrupt")), - ) - db.append_message("s1", "user", "before restart") - db.close() - - raw = sqlite3.connect(str(db_path)) - raw.execute( - "INSERT INTO state_meta (key, value) VALUES (?, ?) " - "ON CONFLICT(key) DO UPDATE SET value = excluded.value", - ( - FTS_REBUILD_DEFERRAL_KEY, - json.dumps({ - "first_seen": 1.0, - "last_seen": 30.0, - "attempts": 2, - "holder_pids": [4242], - }), - ), - ) - raw.commit() - raw.close() - - monkeypatch.setattr( - SessionDB, - "_foreign_state_db_holders", - lambda self: [(9999, str(db_path) + "-wal")], - ) - monkeypatch.setattr( - SessionDB, - "_reap_inactive_orphan_desktop_holders", - lambda self, holders, *, min_age_seconds: [], - ) - monkeypatch.setattr(hermes_state_schema.time, "time", lambda: 120.0) - - reopened = SessionDB(db_path=db_path) - try: - record = json.loads(_meta_value(db_path, FTS_REBUILD_DEFERRAL_KEY)) - assert not _deferral_is_futile(record), record - assert record.get("holder_pids") == [9999] + # Backoff pinned at the cap by the same holder; the holder still there -> no retry. + reopened._fts_stale_retry_after = time.monotonic() + hermes_state_schema._FTS_STALE_RETRY_MAX_SECONDS + reopened._fts_stale_retry_interval = hermes_state_schema._FTS_STALE_RETRY_MAX_SECONDS + assert reopened.retry_deferred_fts_recovery() is False assert reopened._fts_stale is True - finally: - reopened.close() - - def test_orphan_reap_clearing_holders_does_not_mark_futile( - self, db, tmp_path, monkeypatch - ): - """CONTROL: escalate + successful orphan reap stays on the existing rebuild path.""" - if not db._fts_enabled: - pytest.skip("FTS5 unavailable in this build") - db_path = tmp_path / "state.db" - db.create_session("s1", source="test") - db.append_message("s1", "user", "seed") - _corrupt_fts(db_path) - monkeypatch.setattr( - db, - "rebuild_fts", - lambda: (_ for _ in ()).throw(sqlite3.DatabaseError("still corrupt")), - ) - db.append_message("s1", "user", "before restart") - db.close() - - raw = sqlite3.connect(str(db_path)) - raw.execute( - "INSERT INTO state_meta (key, value) VALUES (?, ?) " - "ON CONFLICT(key) DO UPDATE SET value = excluded.value", - ( - FTS_REBUILD_DEFERRAL_KEY, - json.dumps({ - "first_seen": 1.0, - "last_seen": 30.0, - "attempts": 2, - "holder_pids": [4242], - }), - ), - ) - raw.commit() - raw.close() - - holder_scans = iter(([(4242, str(db_path) + "-wal")], [])) - reaped = [] - monkeypatch.setattr( - SessionDB, - "_foreign_state_db_holders", - lambda self: next(holder_scans), - ) - monkeypatch.setattr( - SessionDB, - "_reap_inactive_orphan_desktop_holders", - lambda self, holders, *, min_age_seconds: reaped.extend(holders) or [4242], - ) - monkeypatch.setattr(hermes_state_schema.time, "time", lambda: 120.0) - - reopened = SessionDB(db_path=db_path) - try: - assert reaped == [(4242, str(db_path) + "-wal")] - leftover = _meta_value(db_path, FTS_REBUILD_DEFERRAL_KEY) - if leftover is not None: - assert not _deferral_is_futile(json.loads(leftover)), leftover - assert reopened._fts_stale is False - assert _meta_value(db_path, FTS_STALE_KEY) is None - finally: - reopened.close() - - def test_permanent_holder_clear_resets_stale_retry_backoff( - self, db, tmp_path, monkeypatch - ): - if not db._fts_enabled: - pytest.skip("FTS5 unavailable in this build") - db_path = tmp_path / "state.db" - db.create_session("s1", source="test") - db.append_message("s1", "user", "seed") - _corrupt_fts(db_path) - monkeypatch.setattr( - db, - "rebuild_fts", - lambda: (_ for _ in ()).throw(sqlite3.DatabaseError("still corrupt")), - ) - db.append_message("s1", "user", "before restart") - db.close() - - raw = sqlite3.connect(str(db_path)) - raw.execute( - "INSERT INTO state_meta (key, value) VALUES (?, ?) " - "ON CONFLICT(key) DO UPDATE SET value = excluded.value", - ( - FTS_REBUILD_DEFERRAL_KEY, - json.dumps({ - "first_seen": 1.0, - "last_seen": 30.0, - "attempts": 2, - "holder_pids": [4242], - }), - ), - ) - raw.commit() - raw.close() - - current_holders = [(4242, str(db_path) + "-wal")] - monkeypatch.setattr( - SessionDB, - "_foreign_state_db_holders", - lambda self: list(current_holders), - ) - monkeypatch.setattr( - SessionDB, - "_reap_inactive_orphan_desktop_holders", - lambda self, holders, *, min_age_seconds: [], - ) - monkeypatch.setattr(hermes_state_schema.time, "time", lambda: 120.0) - - reopened = SessionDB(db_path=db_path) - try: - record = json.loads(_meta_value(db_path, FTS_REBUILD_DEFERRAL_KEY)) - assert _deferral_is_futile(record), record - current_holders = [] - reopened._fts_stale_retry_after = time.monotonic() + 3600.0 - reopened._fts_stale_retry_interval = 3600.0 + # Holder leaves: the very next tick retries and rebuilds instead of waiting out the cap. + holders.clear() assert reopened.retry_deferred_fts_recovery() is True assert reopened._fts_stale is False + assert _meta_value(db_path, FTS_REBUILD_DEFERRAL_KEY) is None + assert reopened.search_messages("before restart") finally: reopened.close() diff --git a/tests/test_state_db_stats.py b/tests/test_state_db_stats.py index 27f05f310f..75f7b61ae4 100644 --- a/tests/test_state_db_stats.py +++ b/tests/test_state_db_stats.py @@ -109,32 +109,6 @@ def test_collect_and_render_stale_fts_holder_deferral(populated_db): assert any("4242" in warning and "optimize-storage" in warning for warning in warnings) -def test_render_permanent_holder_deferral_is_actionable(): - from hermes_cli.doctor_state import _render_state_db_stats - - lines = _render_state_db_stats( - _base_stats( - fts_rebuild_deferral={ - "attempts": 12, - "holder_pids": [4242], - "first_seen": 1.0, - "futile": True, - "kind": "permanent_holder", - } - ) - ) - warnings = [ - " ".join((text, detail)).lower() - for kind, text, detail in lines - if kind == "warn" - ] - blob = " ".join(warnings) - assert "4242" in blob - assert "stop the other hermes service" in blob - assert "gateway" in blob - assert "optimize-storage" not in blob or "leave this gateway" in blob - - def test_collect_stats_missing_file_never_raises(tmp_path): stats = collect_state_db_stats(tmp_path / "nope" / "state.db") assert isinstance(stats, dict)