diff --git a/hermes_state_gateway.py b/hermes_state_gateway.py index bbf0f09b81..a54e305052 100644 --- a/hermes_state_gateway.py +++ b/hermes_state_gateway.py @@ -448,24 +448,20 @@ class SessionGatewayMixin: ``ws_orphan_reap`` are recoverable; explicit boundaries (/new, /resume switches, compression splits) are not. - Ordering: rank by ``last_activity_at`` (falling back to ``started_at``) - — ``started_at`` alone resurrected days-old zombie rows over the live - conversation. Rows with messages win, but an empty keyed row is still - returned rather than ``None``: ``None`` mints a brand-new session id, - worse than resuming an empty-but-correctly-keyed row (whose transcript - may live under a compression child). - - Reset fence: an intentional boundary (``session_reset`` or any - non-recoverable end_reason) must block fallback to an *older* row for - the same peer, or the has-messages ranking could reach behind a /new - and restore the exact context the user reset — so a candidate is - rejected when a peer boundary row ended *after* its last activity. + Ranked by ``last_activity_at`` (falling back to ``started_at``) — + ``started_at`` alone resurrected days-old zombie rows. Rows with + messages win, but an empty keyed row is still returned rather than + ``None`` (``None`` mints a brand-new session id; the transcript may + live under a compression child). Reset fence: a candidate is rejected + when a peer boundary row (``session_reset`` or any non-recoverable + end_reason) ended *after* its last activity, or the has-messages + ranking could reach behind a /new and restore the reset context. Fallback for a temporarily-missing exact key still requires the - complete peer tuple (never cross chats/threads/users) and a profile + complete peer tuple (never cross chats/threads/users) plus a profile fence: a Telegram DM's peer tuple is identical for every bot (chat_id == user_id, no thread), so a sibling profile's legacy row would - otherwise be adopted. A row is ours when profile_name is the owner or + otherwise be adopted. A row is ours when profile_name is the owner or NULL; stores outside the profile tree derive no owner and stay unfenced. """ if not session_key: diff --git a/hermes_state_maintenance.py b/hermes_state_maintenance.py index 08ef9db6ab..2ae892a7a4 100644 --- a/hermes_state_maintenance.py +++ b/hermes_state_maintenance.py @@ -118,17 +118,15 @@ class SessionMaintenanceMixin: The TUI/desktop gateway reaps disconnected sessions with an in-process grace timer; a restart destroys the timer and leaves ``ended_at IS - NULL`` forever. This closes rows for ``sources`` whose ``started_at`` - AND canonical last activity are both older than ``max_idle_seconds``, - with ``end_reason='startup_orphan_reap'``. The separate ``started_at`` + NULL`` forever. Closes rows for ``sources`` whose ``started_at`` AND + canonical last activity are both older than ``max_idle_seconds`` with + ``end_reason='startup_orphan_reap'`` (the separate ``started_at`` predicate protects fresh compression/branch children whose copied - activity is old. - - Only pass sources whose lifecycle the caller owns — never messaging - platforms like ``telegram`` (ending those triggers a routing loop). - ``exclude_ids`` spares rows this process still holds in memory. - Non-destructive: messages are kept and the row stays resumable; - first-reason-wins via ``ended_at IS NULL``. + activity is old). Only pass sources whose lifecycle the caller owns — + never messaging platforms like ``telegram`` (ending those triggers a + routing loop). ``exclude_ids`` spares rows this process still holds in + memory. Non-destructive: messages are kept and the row stays + resumable; first-reason-wins via ``ended_at IS NULL``. Cross-backend liveness: with ``respect_gateway_heartbeats``, a row is reaped only when stale AND no live backend (heartbeat within @@ -140,8 +138,7 @@ class SessionMaintenanceMixin: Disable the gate only for sources owned by state.db itself. SELECT, live-lease validation and UPDATE run in one ``BEGIN IMMEDIATE`` - transaction. Active turn leases / compression locks spare the row; - expired guards are removed so their former owner is fenced. + transaction; active turn leases / compression locks spare the row. """ from hermes_state import SessionCompressionInProgressError, SessionTurnLeaseLostError srcs = tuple(s for s in sources if s) diff --git a/hermes_state_repair.py b/hermes_state_repair.py index 183e8e46a9..8033b9cff8 100644 --- a/hermes_state_repair.py +++ b/hermes_state_repair.py @@ -617,7 +617,7 @@ def _backup_db_file(db_path: Path) -> "Tuple[Optional[Path], Optional[str]]": Dedupe: if the newest existing backup is byte-identical to the current recovery image (``_backup_content_identity`` — NOT mtime, NOT ``_db_fingerprint``), reuse it; a repair loop once copied the same damaged - bytes on every restart. The copy lands under a staging name OUTSIDE the + bytes on every restart. The copy lands under a staging name OUTSIDE the ``.malformed-backup-`` prefix: a staging name inside it counts as a backup, sorts NEWEST (prune kept partials and deleted intact copies) and dedupe could return it with no real forensic copy on disk. @@ -1022,16 +1022,13 @@ def repair_state_db_schema(db_path: Path, *, backup: bool = True) -> Dict[str, A indexes reject writes. Two corruption classes: malformed schema / "duplicate object definition" - (even ``PRAGMA`` fails), and FTS write-corruption (base tables read fine, - ``integrity_check`` passes, writes fail through ``messages_fts*`` - triggers). Least-destructive first: (1) rebuild FTS in place via FTS5 - ``'rebuild'``; (2) de-duplicate ``sqlite_master`` (lowest rowid per - ``type``/``name``), FTS preserved; (3) drop the FTS schema + ``VACUUM``, - rebuilt on the next ``SessionDB()`` open. Canonical rows are never - modified by a failed attempt: strategies run on a complete SQLite snapshot - and a successful result is copied back transactionally. A raw backup is - taken first unless ``backup=False``. Surgery is serialised across - processes (:func:`_cross_process_repair_lock`): the gateway, Desktop + (even ``PRAGMA`` fails), and FTS write-corruption (reads and + ``integrity_check`` pass, writes fail through ``messages_fts*`` triggers). + Strategies run least-destructive first (see ``_REPAIR_STRATEGIES``) on a + complete SQLite snapshot; a successful result is copied back + transactionally, so canonical rows are never modified by a failed attempt. + A raw backup is taken first unless ``backup=False``. Surgery is serialised + across processes (:func:`_cross_process_repair_lock`): the gateway, Desktop backend and CLI all open the same file, and concurrent ``writable_schema`` surgery is itself a corruption source. diff --git a/hermes_state_wal.py b/hermes_state_wal.py index 89d3e9c6ec..7d9eda6c7a 100644 --- a/hermes_state_wal.py +++ b/hermes_state_wal.py @@ -252,28 +252,24 @@ def apply_wal_with_fallback( """Set ``journal_mode=WAL`` on ``conn``, falling back to DELETE on failure. Returns the mode actually set (``"wal"`` or ``"delete"``). Shared by - :class:`SessionDB` and ``hermes_cli.kanban_db.connect``. + :class:`SessionDB` and ``hermes_cli.kanban_db.connect``. On + WAL-incompatible filesystems SQLite either raises ``OperationalError`` + ("locking protocol" / "disk I/O error") or — macOS NFS / SMB / AgentFS — + silently refuses and leaves the DB in DELETE; either way we log at ERROR + (once per process per ``db_label``) and fall back to DELETE. + ``require_wal=True`` raises :class:`WalUnsupportedError` instead. - On WAL-incompatible filesystems SQLite either raises ``OperationalError`` - ("locking protocol" / "disk I/O error") or — macOS NFS / SMB / AgentFS NFS - overlay — silently refuses and leaves the DB in DELETE. Either way we log - at ERROR (a write now blocks readers) and fall back to DELETE so the - feature keeps working. ``require_wal=True`` raises - :class:`WalUnsupportedError` instead. - - On SQLite builds with the WAL-reset bug (https://sqlite.org/wal.html#walresetbug, - fixed 3.51.3+, backports 3.50.7 / 3.44.6), refuse to enable WAL on - fresh / non-WAL databases; an already-WAL DB keeps WAL with a warning. - This gate is deliberately RETAINED: an attempt to revert it was confounded - by a newer SQLite; re-measured on the bundled 3.50.4, WAL and DELETE are - both clean, so there is no evidence WAL is safer. + WAL-reset-bug builds (https://sqlite.org/wal.html#walresetbug, fixed + 3.51.3+, backports 3.50.7 / 3.44.6) never enable WAL on fresh / non-WAL + databases; an already-WAL DB keeps WAL with a warning. This gate is + deliberately RETAINED: the attempt to revert it was confounded by a newer + SQLite, and re-measured on the bundled 3.50.4 there is no evidence WAL is + safer. Invariant on every path: never downgrade to DELETE if the on-disk header - reports WAL or the mode cannot be read (see _on_disk_journal_mode). Other - gateway/cron/worker connections may hold the DB open, and a live downgrade - destroys their committed-but-uncheckpointed transactions. - - The ERROR is deduplicated per ``db_label``: once per process per DB. + reports WAL or the mode cannot be read — other gateway/cron/worker + connections may hold the DB open, and a live downgrade destroys their + committed-but-uncheckpointed transactions. """ from hermes_state import is_sqlite_wal_reset_vulnerable, resolve_journal_mode configured = resolve_journal_mode()