fix(gateway): hold startup-watchdog progress leases across state.db auto-maintenance

Construction-time maybe_auto_archive / maybe_auto_prune_and_vacuum ran
synchronously with no report_startup_progress lease. A multi-minute
VACUUM of a large state.db accrues near-zero CPU, so the startup
watchdog misread it as a parked deadlock and killed the attempt with
exit 75, live-locking gateway restarts. Renew the lease per long step
(prune, orphan sweep, VACUUM, archive) since leases clamp at 900 s,
plus one lease around the gateway maintenance block.

Fixes #111092
This commit is contained in:
Kevin Rajan
2026-09-14 13:55:24 -05:00
committed by Teknium
parent 5358780264
commit cef69276ae
4 changed files with 94 additions and 0 deletions

View File

@@ -3602,6 +3602,12 @@ class GatewayRunner(
if self._session_db is not None:
try:
from hermes_cli.config import load_config as _load_full_config
from hermes_startup_watchdog import report_startup_progress as _report_startup_progress
# Startup-watchdog lease (#111092): construction-time maintenance is I/O-bound
# with near-zero CPU, which the watchdog misreads as a parked deadlock. The
# maintenance functions renew per long step; this covers the block as a whole.
# No-op when the watchdog is not armed; never raises.
_report_startup_progress(900.0, phase="gateway_startup_state_maintenance")
_sess_cfg = (_load_full_config().get("sessions") or {})
if _sess_cfg.get("auto_archive", False):
self._session_db._db.maybe_auto_archive(

View File

@@ -10,6 +10,7 @@ from typing import Any, Dict, List, Optional, Tuple
from hermes_state_common import (
AUTO_VACUUM_MIN_FREELIST_RATIO, _id_chunks, _placeholders, _sql_session_last_active, escape_like as _escape_like
)
from hermes_startup_watchdog import report_startup_progress
# caplog tests pin the "hermes_state" logger name.
logger = logging.getLogger("hermes_state")
@@ -397,8 +398,14 @@ class SessionMaintenanceMixin:
result["skipped"] = True
return result
# Prune first: orphans closed below get a full retention window.
# Startup-watchdog leases: each long step is I/O-bound (near-zero CPU), which the
# watchdog's CPU fallback misreads as a parked deadlock. Leases are clamped to
# _MAX_LEASE_S=900 per call, so a multi-minute step renews per step rather than
# once at entry. No-op when the watchdog is not armed; never raises.
report_startup_progress(900.0, phase="state_db_auto_prune")
result["pruned"] = pruned = self.prune_sessions(
older_than_days=retention_days, sessions_dir=sessions_dir, exclude_active_write_guards=True)
report_startup_progress(900.0, phase="state_db_auto_sweep")
closed = self.sweep_orphaned_sessions(
max_idle_seconds=float(retention_days) * 86400.0,
sources=self._AUTO_PRUNE_STALE_OPEN_SOURCES, exclude_pinned=True,
@@ -413,6 +420,9 @@ class SessionMaintenanceMixin:
result["freelist_ratio"] = ratio = self._freelist_ratio()
if ratio is None or ratio > min_vacuum_freelist_ratio:
try:
# VACUUM rewrites every page with ~zero CPU: renew the lease here so
# a multi-minute rewrite on a large state.db never outlives the clamp.
report_startup_progress(900.0, phase="state_db_auto_vacuum")
self.vacuum()
result["vacuumed"] = True
self.set_meta("last_vacuum", str(now))

View File

@@ -13,6 +13,7 @@ from typing import Any, Callable, Dict, List, Optional, Tuple
from agent.session_activity import (
ActivityProvenance, bound_activity_description, normalize_activity_provenance,
)
from hermes_startup_watchdog import report_startup_progress
from hermes_state_common import (
_LISTABLE_CHILD_SQL, _PREVIEW_ELIGIBLE_SQL, _PREVIEW_RAW_SELECT, _RECOVERABLE_END_REASONS,
_RECOVERABLE_END_REASONS_SQL, _RESET_END_REASONS, _legacy_reset_child_sql, _shape_preview,
@@ -1614,6 +1615,10 @@ class SessionSessionsMixin:
if last and now - last < min_interval_hours * 3600:
result["skipped"] = True
return result
# Startup-watchdog lease: the archive sweep is I/O-bound (near-zero CPU),
# which the watchdog's CPU fallback misreads as a parked deadlock.
# No-op when the watchdog is not armed; never raises.
report_startup_progress(900.0, phase="state_db_auto_archive")
archived = result["archived"] = self.archive_stale_sessions(idle_days, exclude_pinned=exclude_pinned)
# Record even a zero-archive run so we don't re-sweep every call.
self.set_meta("last_auto_archive", str(now))

View File

@@ -0,0 +1,73 @@
"""Startup-watchdog progress leases across state.db auto-maintenance.
Regression for #111092: gateway restart live-lock on large installs. The construction-time
maintenance block (auto-archive, auto-prune, orphan sweep, VACUUM) is I/O-bound with
near-zero CPU, which the startup watchdog misreads as a parked deadlock and kills with
exit 75. Each long synchronous step must hold a ``report_startup_progress`` lease; leases
are clamped to 900 s per call, so the lease is renewed per step rather than once at
entry.
Invariant (not a change-detector): every long maintenance step reports a progress lease.
"""
from __future__ import annotations
from pathlib import Path
from hermes_state import SessionDB
import hermes_state_maintenance
import hermes_state_sessions
def _make_db(tmp_path: Path) -> SessionDB:
return SessionDB(db_path=tmp_path / "state.db")
def test_prune_sweep_and_vacuum_each_hold_a_lease(tmp_path, monkeypatch):
leases: list[str] = []
monkeypatch.setattr(
hermes_state_maintenance,
"report_startup_progress",
lambda expected_s, phase="": leases.append(phase),
)
db = _make_db(tmp_path)
try:
# Simulate a large-DB sweep where every step does long I/O-bound work.
monkeypatch.setattr(db, "prune_sessions", lambda **kw: 120)
monkeypatch.setattr(db, "sweep_orphaned_sessions", lambda **kw: [])
monkeypatch.setattr(db, "_freelist_ratio", lambda: 1.0)
monkeypatch.setattr(db, "vacuum", lambda: 0)
result = db.maybe_auto_prune_and_vacuum(
retention_days=90,
min_interval_hours=0,
vacuum=True,
min_vacuum_interval_days=0,
)
assert result["pruned"] == 120
assert result["vacuumed"] is True
# A lease must be held across EACH long step: without them the watchdog
# fires exit 75 mid-sweep on a large state.db (#111092).
assert "state_db_auto_prune" in leases
assert "state_db_auto_sweep" in leases
assert "state_db_auto_vacuum" in leases
finally:
db.close()
def test_auto_archive_holds_a_lease(tmp_path, monkeypatch):
leases: list[str] = []
monkeypatch.setattr(
hermes_state_sessions,
"report_startup_progress",
lambda expected_s, phase="": leases.append(phase),
)
db = _make_db(tmp_path)
try:
monkeypatch.setattr(db, "archive_stale_sessions", lambda *a, **kw: 7)
result = db.maybe_auto_archive(idle_days=3, min_interval_hours=0)
assert result["archived"] == 7
assert "state_db_auto_archive" in leases
finally:
db.close()