fix(state): trim the futile-holder FTS diagnostic to shape and fix the remedy text

Slim redo of the mechanism from #106410 on top of its pick (no wrappers, no
persisted "kind" enum, no process-local flag that dies with the process):

- Futility = the SAME holder PID set has blocked >= _FTS_HOLDER_FUTILE_ATTEMPTS
  (10) deferrals over >= _FTS_HOLDER_FUTILE_SECONDS (30 min); tracked as
  holders_since/holders_attempts in the persisted fts_rebuild_deferral record
  and reset whenever the holder set changes. The 3-deferral/60 s escalate
  window is the orphan-reap gate and stays as is.
- ONE escalated ERROR line names each holder pid + cmdline and the remedy that
  can actually be followed from inside a gateway session: stop ONLY the other
  holder; this process's own retry admits the rebuild within 60 s. The old
  "with the gateway stopped" advice was unrunnable from a gateway-hosted
  session (the gateway is the session) and is gone from both log and doctor.
- hermes doctor renders the futile record distinctly.
- retry_deferred_fts_recovery: a capped backoff earned by holder set X no
  longer applies once the live holder set differs from X, so stopping the
  other service is followed by a retry on the next tick, not up to an hour
  later (the issue's 16-min wait).
- Tests trimmed from 5 to 2 invariants (futile line + doctor entry after N
  same-holder deferrals; backoff reset when the holder set changes); the
  contributor's control tests for changing PIDs / orphan reap are covered by
  the existing test_repeated_deferrals_reap_inactive_orphan_then_rebuild.

The "canonical writes and LIKE search remain available" WARNING is kept
because it is true on origin/main: a stale open drops every FTS trigger, so
the messages INSERT succeeds (probed live with a real state.db + a second
process holding it). Writes fail only when a peer re-publishes triggers over
the corrupt index — a separate class, not this diagnostic.

Refs #106393
This commit is contained in:
teknium1
2026-09-09 05:31:07 -07:00
committed by Teknium
parent e7ca9b47ac
commit 060cd7f9bb
4 changed files with 139 additions and 332 deletions

View File

@@ -80,21 +80,17 @@ def _render_state_db_stats(stats: dict, holders=None) -> list:
lines.append(("info", "FTS tables: " + (", ".join(present) if present else "none"), ""))
deferral = stats.get("fts_rebuild_deferral")
if isinstance(deferral, dict):
pids = deferral.get("holder_pids") or []
attempts = deferral.get("attempts") or "?"
futile = deferral.get("futile") is True or deferral.get("kind") == "permanent_holder"
if futile:
lines.append((
"warn",
f"state.db FTS repair is blocked by a permanent holder "
f"(another Hermes service, PID(s) {pids}) after {attempts} futile deferral(s)",
"(stop the other Hermes service; leave this gateway running — "
"retry_deferred_fts_recovery admits once this process is the sole holder)",
))
pids = deferral.get("holder_pids") or "unknown"
if deferral.get("futile"):
lines.append(("warn", f"state.db FTS repair is blocked by the same holder(s) PID(s) {pids} for "
f"{deferral.get('holders_attempts') or '?'} consecutive deferral(s); waiting is futile",
"(stop ONLY the listed process(es) — the gateway keeps running and its own retry "
"rebuilds within a minute of the holder leaving)"))
else:
lines.append(("warn", f"state.db FTS repair is blocked after {attempts} deferral(s) "
f"by PID(s) {pids or 'unknown'}",
"(stop the listed processes, then run 'hermes sessions optimize-storage' with the gateway stopped)"))
lines.append(("warn", f"state.db FTS repair is blocked after {deferral.get('attempts') or '?'} deferral(s) "
f"by PID(s) {pids}",
"(stop the listed processes; the gateway's own retry then rebuilds, or run "
"'hermes sessions optimize-storage' with every holder stopped)"))
# Oversized DB: suggest auto_prune, plus the offline optimize-storage pass when the FTS rebuild is
# pending OR the DB predates the current trigram layout (fts_storage_version < FTS_STORAGE_VERSION).
if logical is not None and logical > STATE_DB_SIZE_WARN_BYTES:

View File

@@ -26,31 +26,26 @@ from hermes_state_common import (
LEGACY_FTS_TRIGRAM_SQL, SCHEMA_SQL,
SCHEMA_VERSION, _FTS_CJK_TRIGGERS, _FTS_TRIGGERS, _ephemeral_child_sql, fts_rebuild_admission,
)
from hermes_state_holders import _read_proc_argv
# Pre-split logger identity so log filtering/capture is unchanged.
logger = logging.getLogger("hermes_state")
_FTS_HOLDER_ESCALATE_ATTEMPTS = 3
_FTS_HOLDER_ESCALATE_SECONDS = 60.0
# The same holder PID set blocking this many deferrals over this long is a structurally resident
# peer (a supervised service on the same HERMES_HOME), not a transient one worth waiting out (#106393).
_FTS_HOLDER_FUTILE_ATTEMPTS = 10
_FTS_HOLDER_FUTILE_SECONDS = 1800.0
# retry_deferred_fts_recovery cadence: startup paid the full admission wait once; later
# retries are non-blocking probes whose spacing doubles up to the cap.
_FTS_STALE_RETRY_SECONDS = 60.0
_FTS_STALE_RETRY_MAX_SECONDS = 3600.0
def _fts_holder_pid_set(value):
"""Sorted unique positive PIDs, or None when the stored set is missing/unusable."""
if not isinstance(value, (list, tuple)):
return None
pids = set()
for item in value:
try:
pid = int(item)
except (TypeError, ValueError):
return None
if pid > 0:
pids.add(pid)
return sorted(pids)
def _holder_cmdline(pid: int) -> str:
argv = _read_proc_argv(pid)
return " ".join(argv)[:120] if argv else "<cmdline unavailable>"
# schema_read_probe_statements() cache (parses SCHEMA_SQL in an in-memory DB; once per process).
_READ_PROBE_STATEMENTS: Optional[tuple] = None
@@ -437,8 +432,12 @@ class SessionSchemaMixin:
"""Record a deferral diagnostic for the foreign processes holding the DB; True = defer
(holders remain). After ``_FTS_HOLDER_ESCALATE_ATTEMPTS`` deferrals spanning
``_FTS_HOLDER_ESCALATE_SECONDS``, provably inactive orphan Desktop backends are
reaped and the holders re-checked. A stable PID set that survives that reap is
recorded as a futile/permanent-holder diagnostic; the orphan predicate is unchanged."""
reaped and the holders re-checked. The orphan reap is the only exit, so a supervised
peer (never an orphan) blocks forever: once the SAME PID set has blocked
``_FTS_HOLDER_FUTILE_ATTEMPTS`` deferrals over ``_FTS_HOLDER_FUTILE_SECONDS`` the
record is marked ``futile`` and the escalation names the holders and the remedy that
works from inside a gateway session (stop only the other holder; this process's own
retry tick admits the rebuild). A changed holder set restarts that window."""
now = time.time()
try:
row = cursor.execute(
@@ -451,13 +450,15 @@ class SessionSchemaMixin:
try:
first_seen = float(record.get("first_seen", now))
attempts = int(record.get("attempts", 0)) + 1
holders_since = float(record.get("holders_since", now))
holders_attempts = int(record.get("holders_attempts", 0)) + 1
except (TypeError, ValueError):
first_seen, attempts = now, 1
first_seen, attempts, holders_since, holders_attempts = now, 1, now, 1
if first_seen > now or first_seen < 0:
first_seen = now
previous_pids = _fts_holder_pid_set(record.get("holder_pids"))
holder_pids = sorted({pid for pid, _path in foreign_holders if pid > 0})
futile = False
if holder_pids != record.get("holder_pids"):
holders_since, holders_attempts = now, 1
if attempts >= _FTS_HOLDER_ESCALATE_ATTEMPTS and now - first_seen >= _FTS_HOLDER_ESCALATE_SECONDS:
reaped = self._reap_inactive_orphan_desktop_holders(
foreign_holders, min_age_seconds=_FTS_HOLDER_ESCALATE_SECONDS,
@@ -469,58 +470,40 @@ class SessionSchemaMixin:
)
foreign_holders = self._foreign_state_db_holders()
holder_pids = sorted({pid for pid, _path in foreign_holders if pid > 0})
if foreign_holders:
if previous_pids is not None and holder_pids and holder_pids == previous_pids:
futile = True
logger.error(
"state.db FTS repair is futile after %d deferrals: the same "
"holder PID set %s is a permanent holder (another Hermes "
"service), not a transient peer. Stop the other Hermes "
"service; leave this gateway running. "
"retry_deferred_fts_recovery admits once this process is "
"the sole holder. `hermes doctor` reports this degraded state.",
attempts, holder_pids,
)
else:
logger.error(
"state.db FTS repair remains blocked after %d deferrals "
"by holder(s) %s. Stop the listed processes, then run "
"`hermes sessions optimize-storage` with the gateway stopped. "
"`hermes doctor` reports this degraded state.", attempts, foreign_holders,
)
else:
self._fts_stale_retry_after = 0.0
self._fts_stale_retry_interval = 0.0
self._fts_permanent_holder_deferral = False
futile = bool(holder_pids) and (
holders_attempts >= _FTS_HOLDER_FUTILE_ATTEMPTS and now - holders_since >= _FTS_HOLDER_FUTILE_SECONDS
)
diagnostic = {
"first_seen": first_seen, "last_seen": now, "attempts": attempts,
"holder_pids": holder_pids,
"first_seen": first_seen, "last_seen": now, "attempts": attempts, "holder_pids": holder_pids,
"holders_since": holders_since, "holders_attempts": holders_attempts, "futile": futile,
}
if futile:
diagnostic["futile"] = True
diagnostic["kind"] = "permanent_holder"
self._fts_permanent_holder_deferral = True
else:
self._fts_permanent_holder_deferral = False
cursor.execute(
"INSERT INTO state_meta (key, value) VALUES (?, ?) ON CONFLICT(key) DO UPDATE SET value = excluded.value",
(FTS_REBUILD_DEFERRAL_KEY, json.dumps(diagnostic, sort_keys=True)),
)
if not foreign_holders:
self._fts_deferred_holder_pids = None
return False
self._fts_deferred_holder_pids = holder_pids
if futile:
logger.warning(
"Deferred stale state.db FTS rebuild while a permanent foreign holder "
"blocks repair (%s); stop the other Hermes service and leave this "
"gateway running (deferral %d).",
foreign_holders, attempts,
logger.error(
"state.db FTS repair has been blocked by the same holder(s) for %d deferrals over %.0f min "
"(%s); waiting is futile. Stop ONLY the other holder(s) — this process keeps running and its "
"own retry admits the rebuild within %.0fs of the holder leaving. `hermes doctor` shows this.",
holders_attempts, (now - holders_since) / 60.0,
", ".join(f"pid {pid}: {_holder_cmdline(pid)}" for pid in holder_pids), _FTS_STALE_RETRY_SECONDS,
)
else:
logger.warning(
"Deferred stale state.db FTS rebuild while foreign processes "
"hold the database or WAL sidecars (%s); canonical writes and LIKE search remain available (deferral %d).",
foreign_holders, attempts,
elif attempts >= _FTS_HOLDER_ESCALATE_ATTEMPTS and now - first_seen >= _FTS_HOLDER_ESCALATE_SECONDS:
logger.error(
"state.db FTS repair remains blocked after %d deferrals by holder(s) %s. Stop the listed "
"processes (this process's own retry then rebuilds), or run `hermes sessions optimize-storage` "
"with every holder stopped. `hermes doctor` reports this degraded state.", attempts, foreign_holders,
)
logger.warning(
"Deferred stale state.db FTS rebuild while foreign processes "
"hold the database or WAL sidecars (%s); canonical writes and LIKE search remain available (deferral %d).",
foreign_holders, attempts,
)
return True
def _recover_stale_fts(self, cursor: sqlite3.Cursor, *, legacy: bool, timeout_seconds=None) -> bool:
@@ -563,13 +546,15 @@ class SessionSchemaMixin:
if self.read_only or self._conn is None:
return False
now = time.monotonic()
if getattr(self, "_fts_permanent_holder_deferral", False) and not self._foreign_state_db_holders():
# Permanent holder gone: do not wait out a doubled-to-cap backoff (#106393).
self._fts_stale_retry_after = 0.0
self._fts_stale_retry_interval = 0.0
self._fts_permanent_holder_deferral = False
deferred_pids = getattr(self, "_fts_deferred_holder_pids", None)
if now < getattr(self, "_fts_stale_retry_after", 0.0):
return False
# The backoff was earned by a specific holder set; once that set changes (the other
# service stopped) a capped backoff would idle up to an hour with nothing blocking (#106393).
if deferred_pids is None or sorted(
{pid for pid, _path in self._foreign_state_db_holders() if pid > 0}
) == deferred_pids:
return False
self._fts_stale_retry_interval = 0.0
interval = float(getattr(self, "_fts_stale_retry_interval", 0.0))
if interval <= 0.0:
interval = _FTS_STALE_RETRY_SECONDS

View File

@@ -14,6 +14,7 @@ rebuild later, outside the failed live write/search operation.
"""
import json
import logging
import os
import sqlite3
import time
@@ -28,24 +29,6 @@ from hermes_state_common import FTS_REBUILD_DEFERRAL_KEY, FTS_STALE_KEY, LEGACY_
from hermes_state_dbfile import _concrete_state_db_holder_pids, _is_inactive_orphan_desktop_holder
def _deferral_is_futile(record):
if not isinstance(record, dict):
return False
if record.get("futile") is True:
return True
kind = str(record.get("kind") or record.get("status") or "").lower()
return "futile" in kind or "permanent_holder" in kind
def _doctor_deferral_blob(stats):
from hermes_cli.doctor_state import _render_state_db_stats
return " ".join(
" ".join(str(part) for part in row)
for row in _render_state_db_stats(stats)
).lower()
@pytest.fixture
def db(tmp_path):
d = SessionDB(db_path=tmp_path / "state.db")
@@ -804,10 +787,13 @@ class TestRuntimeFtsRebuild:
finally:
reopened.close()
def test_stable_permanent_holder_marks_futile_deferral(
def test_same_holder_set_across_futile_window_names_holder_and_gateway_safe_remedy(
self, db, tmp_path, monkeypatch, caplog
):
"""Same supervised holder across the escalate window is futile, not a transient peer."""
"""#106393: a supervised peer never satisfies the orphan reap, so the generic
'remains blocked ... with the gateway stopped' escalation repeats forever. Once the SAME
PID set has blocked the futile window, the record is marked futile, the escalation names
the holder's cmdline and a remedy runnable from inside the gateway, and doctor says so."""
if not db._fts_enabled:
pytest.skip("FTS5 unavailable in this build")
db_path = tmp_path / "state.db"
@@ -815,229 +801,95 @@ class TestRuntimeFtsRebuild:
db.append_message("s1", "user", "seed")
_corrupt_fts(db_path)
monkeypatch.setattr(
db,
"rebuild_fts",
lambda: (_ for _ in ()).throw(sqlite3.DatabaseError("still corrupt")),
db, "rebuild_fts", lambda: (_ for _ in ()).throw(sqlite3.DatabaseError("still corrupt")),
)
db.append_message("s1", "user", "before restart")
db.close()
raw = sqlite3.connect(str(db_path))
raw.execute(
"INSERT INTO state_meta (key, value) VALUES (?, ?) "
"ON CONFLICT(key) DO UPDATE SET value = excluded.value",
(
FTS_REBUILD_DEFERRAL_KEY,
json.dumps({
"first_seen": 1.0,
"last_seen": 30.0,
"attempts": 2,
"holder_pids": [4242],
}),
),
)
raw.commit()
raw.close()
monkeypatch.setattr(
SessionDB,
"_foreign_state_db_holders",
lambda self: [(4242, str(db_path) + "-wal")],
SessionDB, "_foreign_state_db_holders", lambda self: [(4242, str(db_path) + "-wal")],
)
monkeypatch.setattr(
SessionDB,
"_reap_inactive_orphan_desktop_holders",
lambda self, holders, *, min_age_seconds: [],
SessionDB, "_reap_inactive_orphan_desktop_holders", lambda self, holders, *, min_age_seconds: [],
)
monkeypatch.setattr(hermes_state_schema.time, "time", lambda: 120.0)
monkeypatch.setattr(
hermes_state_schema, "_read_proc_argv", lambda pid: ["python", "-m", "hermes_cli.main", "serve"], raising=False,
)
clock = [1000.0]
monkeypatch.setattr(hermes_state_schema.time, "time", lambda: clock[0])
from hermes_cli.doctor_state import _render_state_db_stats
from hermes_state_dbfile import collect_state_db_stats
def doctor_blob():
return " ".join(" ".join(row) for row in _render_state_db_stats(collect_state_db_stats(db_path))).lower()
# Read via getattr so the red-on-base run reaches the behavioural assertion, not a NameError.
futile_attempts = getattr(hermes_state_schema, "_FTS_HOLDER_FUTILE_ATTEMPTS", 10)
futile_seconds = getattr(hermes_state_schema, "_FTS_HOLDER_FUTILE_SECONDS", 1800.0)
reopened = SessionDB(db_path=db_path)
try:
cursor = reopened._conn.cursor()
# One deferral short of the futile window: still the generic escalation.
for _ in range(futile_attempts - 2):
clock[0] += futile_seconds
caplog.clear()
assert reopened._recover_stale_fts(cursor, legacy=False, timeout_seconds=0.0) is False
reopened._conn.commit()
assert not json.loads(_meta_value(db_path, FTS_REBUILD_DEFERRAL_KEY)).get("futile")
assert "waiting is futile" not in caplog.text
assert "futile" not in doctor_blob()
clock[0] += futile_seconds
caplog.clear()
assert reopened._recover_stale_fts(cursor, legacy=False, timeout_seconds=0.0) is False
reopened._conn.commit()
record = json.loads(_meta_value(db_path, FTS_REBUILD_DEFERRAL_KEY))
assert _deferral_is_futile(record), record
assert record.get("holder_pids") == [4242]
assert record.get("futile") is True and record["holder_pids"] == [4242]
futile_lines = [r for r in caplog.records if "waiting is futile" in r.getMessage()]
assert len(futile_lines) == 1 and futile_lines[0].levelno == logging.ERROR
msg = futile_lines[0].getMessage()
assert "pid 4242: python -m hermes_cli.main serve" in msg
assert "Stop ONLY the other holder" in msg and "with the gateway stopped" not in msg
blob = doctor_blob()
assert "4242" in blob and "waiting is futile" in blob and "stop only" in blob
assert "gateway stopped" not in blob
finally:
reopened.close()
def test_retry_backoff_resets_when_the_blocking_holder_set_changes(
self, db, tmp_path, monkeypatch
):
"""#106393: days of deferrals pin the retry interval at the 1h cap, so stopping the other
holder was followed by up to an hour of nothing. A changed holder set must retry now."""
if not db._fts_enabled:
pytest.skip("FTS5 unavailable in this build")
db_path = tmp_path / "state.db"
db.create_session("s1", source="test")
db.append_message("s1", "user", "seed")
_corrupt_fts(db_path)
monkeypatch.setattr(
db, "rebuild_fts", lambda: (_ for _ in ()).throw(sqlite3.DatabaseError("still corrupt")),
)
db.append_message("s1", "user", "before restart")
db.close()
holders = [(4242, str(db_path) + "-wal")]
monkeypatch.setattr(SessionDB, "_foreign_state_db_holders", lambda self: list(holders))
reopened = SessionDB(db_path=db_path)
try:
assert reopened._fts_stale is True
from hermes_cli.doctor_state import _render_state_db_stats
from hermes_state_dbfile import collect_state_db_stats
blob = _doctor_deferral_blob(collect_state_db_stats(db_path))
assert "stop the other hermes service" in blob
assert "gateway" in blob
assert "canonical writes and like search remain available" not in caplog.text.lower()
finally:
reopened.close()
def test_changing_holder_pids_do_not_mark_futile(
self, db, tmp_path, monkeypatch
):
if not db._fts_enabled:
pytest.skip("FTS5 unavailable in this build")
db_path = tmp_path / "state.db"
db.create_session("s1", source="test")
db.append_message("s1", "user", "seed")
_corrupt_fts(db_path)
monkeypatch.setattr(
db,
"rebuild_fts",
lambda: (_ for _ in ()).throw(sqlite3.DatabaseError("still corrupt")),
)
db.append_message("s1", "user", "before restart")
db.close()
raw = sqlite3.connect(str(db_path))
raw.execute(
"INSERT INTO state_meta (key, value) VALUES (?, ?) "
"ON CONFLICT(key) DO UPDATE SET value = excluded.value",
(
FTS_REBUILD_DEFERRAL_KEY,
json.dumps({
"first_seen": 1.0,
"last_seen": 30.0,
"attempts": 2,
"holder_pids": [4242],
}),
),
)
raw.commit()
raw.close()
monkeypatch.setattr(
SessionDB,
"_foreign_state_db_holders",
lambda self: [(9999, str(db_path) + "-wal")],
)
monkeypatch.setattr(
SessionDB,
"_reap_inactive_orphan_desktop_holders",
lambda self, holders, *, min_age_seconds: [],
)
monkeypatch.setattr(hermes_state_schema.time, "time", lambda: 120.0)
reopened = SessionDB(db_path=db_path)
try:
record = json.loads(_meta_value(db_path, FTS_REBUILD_DEFERRAL_KEY))
assert not _deferral_is_futile(record), record
assert record.get("holder_pids") == [9999]
# Backoff pinned at the cap by the same holder; the holder still there -> no retry.
reopened._fts_stale_retry_after = time.monotonic() + hermes_state_schema._FTS_STALE_RETRY_MAX_SECONDS
reopened._fts_stale_retry_interval = hermes_state_schema._FTS_STALE_RETRY_MAX_SECONDS
assert reopened.retry_deferred_fts_recovery() is False
assert reopened._fts_stale is True
finally:
reopened.close()
def test_orphan_reap_clearing_holders_does_not_mark_futile(
self, db, tmp_path, monkeypatch
):
"""CONTROL: escalate + successful orphan reap stays on the existing rebuild path."""
if not db._fts_enabled:
pytest.skip("FTS5 unavailable in this build")
db_path = tmp_path / "state.db"
db.create_session("s1", source="test")
db.append_message("s1", "user", "seed")
_corrupt_fts(db_path)
monkeypatch.setattr(
db,
"rebuild_fts",
lambda: (_ for _ in ()).throw(sqlite3.DatabaseError("still corrupt")),
)
db.append_message("s1", "user", "before restart")
db.close()
raw = sqlite3.connect(str(db_path))
raw.execute(
"INSERT INTO state_meta (key, value) VALUES (?, ?) "
"ON CONFLICT(key) DO UPDATE SET value = excluded.value",
(
FTS_REBUILD_DEFERRAL_KEY,
json.dumps({
"first_seen": 1.0,
"last_seen": 30.0,
"attempts": 2,
"holder_pids": [4242],
}),
),
)
raw.commit()
raw.close()
holder_scans = iter(([(4242, str(db_path) + "-wal")], []))
reaped = []
monkeypatch.setattr(
SessionDB,
"_foreign_state_db_holders",
lambda self: next(holder_scans),
)
monkeypatch.setattr(
SessionDB,
"_reap_inactive_orphan_desktop_holders",
lambda self, holders, *, min_age_seconds: reaped.extend(holders) or [4242],
)
monkeypatch.setattr(hermes_state_schema.time, "time", lambda: 120.0)
reopened = SessionDB(db_path=db_path)
try:
assert reaped == [(4242, str(db_path) + "-wal")]
leftover = _meta_value(db_path, FTS_REBUILD_DEFERRAL_KEY)
if leftover is not None:
assert not _deferral_is_futile(json.loads(leftover)), leftover
assert reopened._fts_stale is False
assert _meta_value(db_path, FTS_STALE_KEY) is None
finally:
reopened.close()
def test_permanent_holder_clear_resets_stale_retry_backoff(
self, db, tmp_path, monkeypatch
):
if not db._fts_enabled:
pytest.skip("FTS5 unavailable in this build")
db_path = tmp_path / "state.db"
db.create_session("s1", source="test")
db.append_message("s1", "user", "seed")
_corrupt_fts(db_path)
monkeypatch.setattr(
db,
"rebuild_fts",
lambda: (_ for _ in ()).throw(sqlite3.DatabaseError("still corrupt")),
)
db.append_message("s1", "user", "before restart")
db.close()
raw = sqlite3.connect(str(db_path))
raw.execute(
"INSERT INTO state_meta (key, value) VALUES (?, ?) "
"ON CONFLICT(key) DO UPDATE SET value = excluded.value",
(
FTS_REBUILD_DEFERRAL_KEY,
json.dumps({
"first_seen": 1.0,
"last_seen": 30.0,
"attempts": 2,
"holder_pids": [4242],
}),
),
)
raw.commit()
raw.close()
current_holders = [(4242, str(db_path) + "-wal")]
monkeypatch.setattr(
SessionDB,
"_foreign_state_db_holders",
lambda self: list(current_holders),
)
monkeypatch.setattr(
SessionDB,
"_reap_inactive_orphan_desktop_holders",
lambda self, holders, *, min_age_seconds: [],
)
monkeypatch.setattr(hermes_state_schema.time, "time", lambda: 120.0)
reopened = SessionDB(db_path=db_path)
try:
record = json.loads(_meta_value(db_path, FTS_REBUILD_DEFERRAL_KEY))
assert _deferral_is_futile(record), record
current_holders = []
reopened._fts_stale_retry_after = time.monotonic() + 3600.0
reopened._fts_stale_retry_interval = 3600.0
# Holder leaves: the very next tick retries and rebuilds instead of waiting out the cap.
holders.clear()
assert reopened.retry_deferred_fts_recovery() is True
assert reopened._fts_stale is False
assert _meta_value(db_path, FTS_REBUILD_DEFERRAL_KEY) is None
assert reopened.search_messages("before restart")
finally:
reopened.close()

View File

@@ -109,32 +109,6 @@ def test_collect_and_render_stale_fts_holder_deferral(populated_db):
assert any("4242" in warning and "optimize-storage" in warning for warning in warnings)
def test_render_permanent_holder_deferral_is_actionable():
from hermes_cli.doctor_state import _render_state_db_stats
lines = _render_state_db_stats(
_base_stats(
fts_rebuild_deferral={
"attempts": 12,
"holder_pids": [4242],
"first_seen": 1.0,
"futile": True,
"kind": "permanent_holder",
}
)
)
warnings = [
" ".join((text, detail)).lower()
for kind, text, detail in lines
if kind == "warn"
]
blob = " ".join(warnings)
assert "4242" in blob
assert "stop the other hermes service" in blob
assert "gateway" in blob
assert "optimize-storage" not in blob or "leave this gateway" in blob
def test_collect_stats_missing_file_never_raises(tmp_path):
stats = collect_state_db_stats(tmp_path / "nope" / "state.db")
assert isinstance(stats, dict)