From 176458cdd975282d0f4dcf19588111d1af7f6141 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 16:02:42 -0700 Subject: [PATCH] =?UTF-8?q?refactor(state):=20wal=20=E2=80=94=20=5Fwarn=5F?= =?UTF-8?q?once/=5Fmode=5Ffrom=5Frow/=5Fapply=5Fwal=5Fcompanions=20helpers?= =?UTF-8?q?,=20shared=20delete-verify=20message,=20compact=20docs?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- hermes_state_wal.py | 405 +++++++++++++++----------------------------- 1 file changed, 139 insertions(+), 266 deletions(-) diff --git a/hermes_state_wal.py b/hermes_state_wal.py index a6a1ce9171..89d3e9c6ec 100644 --- a/hermes_state_wal.py +++ b/hermes_state_wal.py @@ -30,42 +30,56 @@ logger = logging.getLogger("hermes_state") # filesystems (NFS, SMB/CIFS, some FUSE, WSL1) don't provide reliably — there # ``PRAGMA journal_mode=WAL`` raises ``locking protocol`` (SQLITE_PROTOCOL). # ZFS instead corrupts the -shm file under concurrent connection bursts (COW + -# mmap), presenting as ``disk I/O error``. Propagating either would silently -# break everything backed by state.db/kanban.db, so we fall back to -# ``journal_mode=DELETE`` (pre-WAL default, works on NFS/ZFS): readers block -# during a write, but it works. The WAL-reset-bug gate and the -# never-live-downgrade invariant are documented on apply_wal_with_fallback. +# mmap), presenting as ``disk I/O error``. Either would silently break +# everything backed by state.db/kanban.db, so we fall back to +# ``journal_mode=DELETE`` (works on NFS/ZFS; readers block during a write). _WAL_INCOMPAT_MARKERS = ( "locking protocol", # SQLITE_PROTOCOL on NFS/SMB "not authorized", # Some FUSE mounts block WAL pragma outright "disk i/o error", # ZFS SHM corruption under concurrent connections ) - # SQLite's default is -1 (unlimited), so state.db-wal would keep the high-water # mark of the largest-ever transaction forever. See _apply_wal_size_limit(). _WAL_SIZE_LIMIT_BYTES = 64 * 1024 * 1024 # 64 MiB - -# Dedup sets: kanban_db.connect() runs on every kanban operation, so an -# undeduped fallback log line would repeat per connection and fill errors.log. +# Once-per-process-per-db_label dedup sets: kanban_db.connect() runs on every +# kanban operation, so an undeduped log line would repeat per connection. +# Tests clear these through ``hermes_state.``; ``_warn_once`` resolves +# the set through hermes_state at call time for the same reason. _wal_fallback_warned_paths: set[str] = set() - - _wal_fallback_warned_lock = threading.Lock() - - _wal_reset_bug_warned_paths: set[str] = set() - - _wal_reset_bug_warned_lock = threading.Lock() - - # "configured delete overridden by on-disk WAL" ERROR. _delete_overridden_warned_paths: set[str] = set() - - _delete_overridden_warned_lock = threading.Lock() +# Dedup state for _log_journal_mode_upgrade_once. +_journal_upgrade_warned_paths: set = set() +_journal_upgrade_warned_lock = threading.Lock() + +_CANNOT_VERIFY_DELETE_MSG = ( + "could not verify journal mode before applying configured " + "journal_mode=delete (database is locked — possible " + "concurrent openers); refusing to downgrade a database " + "this process does not exclusively own" +) + + +def _warn_once(lock: threading.Lock, set_name: str, key: str) -> bool: + """True the first time *key* is seen in ``hermes_state.``.""" + import hermes_state + seen = getattr(hermes_state, set_name) + with lock: + if key in seen: + return False + seen.add(key) + return True + + +def _mode_from_row(row) -> str: + """Lower-cased mode from a ``PRAGMA journal_mode`` row, ``""`` if no row.""" + return str(row[0]).strip().lower() if row and row[0] is not None else "" def _on_disk_journal_mode(conn: sqlite3.Connection) -> Optional[str]: @@ -96,9 +110,7 @@ def _on_disk_journal_mode(conn: sqlite3.Connection) -> Optional[str]: return None return str(mode).strip().lower() if mode is not None else None if last_exc is not None: - logger.debug( - "_on_disk_journal_mode: retries exhausted on disk read (%s)", last_exc - ) + logger.debug("_on_disk_journal_mode: retries exhausted on disk read (%s)", last_exc) return None @@ -107,15 +119,10 @@ def _apply_wal_size_limit(conn: sqlite3.Connection) -> None: SQLite's default ``journal_size_limit`` is -1: a checkpointed WAL is reused in place, never truncated, so ``state.db-wal`` keeps the high-water mark - of the largest transaction ever run. One bulk op strands gigabytes — - ``hermes sessions optimize`` on a 3 GB state.db left a 3 GB WAL and filled - the disk, so the maintenance command was self-defeating on the largest - DBs. With a limit, each checkpoint truncates the WAL back to it; 64 MiB is - above normal transaction sizes (steady-state commits never pay a truncate) - while capping slack predictably. kanban_db uses ``wal_autocheckpoint=100``. - - Best-effort: never raises — failure only costs disk slack and must not - prevent the database from opening. + of the largest transaction ever run (a 3 GB optimize left a 3 GB WAL). + With a limit, each checkpoint truncates the WAL back to it; 64 MiB is + above normal transaction sizes while capping slack predictably. + Best-effort: failure only costs disk slack and must not prevent opening. """ try: conn.execute(f"PRAGMA journal_size_limit={_WAL_SIZE_LIMIT_BYTES}") @@ -127,12 +134,11 @@ def _apply_macos_checkpoint_barrier(conn: sqlite3.Connection) -> None: """Enable ``PRAGMA checkpoint_fullfsync`` on macOS (no-op elsewhere). Apple's ``fsync(2)`` guarantees neither data-on-platter nor write ordering, - so WAL's corruption-safety assumption fails on Darwin without ``F_FULLFSYNC``. - A launchd shutdown drops the page cache (power-loss for in-flight pages), so - a checkpoint that "reported" durable can leave a malformed ``state.db``; - a plain in-session kill survives via the page cache. The barrier applies - only at checkpoint boundaries (~+0.1 ms/commit vs ~+4 ms for - ``fullfsync=1``). Best-effort: never raises. + so WAL's corruption-safety assumption fails on Darwin without ``F_FULLFSYNC``: + a launchd shutdown drops the page cache and a checkpoint that "reported" + durable can leave a malformed ``state.db``. The barrier applies only at + checkpoint boundaries (~+0.1 ms/commit vs ~+4 ms for ``fullfsync=1``). + Best-effort: never raises. """ if sys.platform != "darwin": return @@ -145,11 +151,10 @@ def _apply_macos_checkpoint_barrier(conn: sqlite3.Connection) -> None: def _enforce_macos_synchronous_full(conn: sqlite3.Connection) -> None: """Enforce ``PRAGMA synchronous=FULL`` on macOS to prevent btree corruption. - With NORMAL, a WAL checkpoint racing process termination (launchd shutdown) - can leave half-written btree pages (``btreeInitPage error 11``) because - Darwin's ``fsync()`` guarantees neither ordering nor durability. Called - after every successful WAL activation so a prior connection's NORMAL never - sticks. Best-effort: never raises. + With NORMAL, a WAL checkpoint racing process termination can leave + half-written btree pages (``btreeInitPage error 11``). Called after every + successful WAL activation so a prior connection's NORMAL never sticks. + Best-effort: never raises. """ if sys.platform != "darwin": return @@ -159,9 +164,14 @@ def _enforce_macos_synchronous_full(conn: sqlite3.Connection) -> None: pass -def is_sqlite_wal_reset_vulnerable( - version_info: Optional[tuple] = None, -) -> bool: +def _apply_wal_companions(conn: sqlite3.Connection) -> None: + """The settings every WAL activation carries: size limit + macOS barriers.""" + _apply_wal_size_limit(conn) + _apply_macos_checkpoint_barrier(conn) + _enforce_macos_synchronous_full(conn) + + +def is_sqlite_wal_reset_vulnerable(version_info: Optional[tuple] = None) -> bool: """True when the linked SQLite has the WAL-reset bug (3.7.0–3.51.2; fixed 3.51.3+, backports 3.50.7 / 3.44.6). Pre-WAL libraries are safe. https://sqlite.org/wal.html#walresetbug @@ -190,8 +200,7 @@ def _database_has_content(conn: sqlite3.Connection) -> bool: ``PRAGMA page_count`` is a lock-free header read. Fail-quiet: any error answers False, because the only caller gates a warning on this and an - unknown-answer warning would fire on every fresh database — exactly where - there is provably no operator choice being overwritten. + unknown-answer warning would fire on every fresh database. """ try: row = conn.execute("PRAGMA page_count").fetchone() @@ -222,7 +231,6 @@ def resolve_journal_mode() -> str: raw = database.get("journal_mode", "wal") except Exception: return "wal" - if not isinstance(raw, str): return "wal" mode = raw.strip().lower() @@ -239,43 +247,33 @@ class WalUnsupportedError(sqlite3.OperationalError): def apply_wal_with_fallback( - conn: sqlite3.Connection, - *, - db_label: str = "state.db", - require_wal: bool = False, + conn: sqlite3.Connection, *, db_label: str = "state.db", require_wal: bool = False ) -> str: """Set ``journal_mode=WAL`` on ``conn``, falling back to DELETE on failure. Returns the mode actually set (``"wal"`` or ``"delete"``). Shared by - :class:`SessionDB` and ``hermes_cli.kanban_db.connect`` for identical - fallback behavior. + :class:`SessionDB` and ``hermes_cli.kanban_db.connect``. - On WAL-incompatible filesystems (NFS, SMB, some FUSE, ZFS) SQLite either - raises ``OperationalError`` ("locking protocol" / "disk I/O error") or — - macOS NFS / SMB / AgentFS NFS overlay — silently refuses and leaves the DB - in DELETE. Either way we log at ERROR (a write now blocks readers — a real - concurrency loss) and fall back to DELETE so the feature keeps working. - ``require_wal=True`` raises :class:`WalUnsupportedError` instead; all - current callers keep the default so NFS-homed installs work. + On WAL-incompatible filesystems SQLite either raises ``OperationalError`` + ("locking protocol" / "disk I/O error") or — macOS NFS / SMB / AgentFS NFS + overlay — silently refuses and leaves the DB in DELETE. Either way we log + at ERROR (a write now blocks readers) and fall back to DELETE so the + feature keeps working. ``require_wal=True`` raises + :class:`WalUnsupportedError` instead. On SQLite builds with the WAL-reset bug (https://sqlite.org/wal.html#walresetbug, fixed 3.51.3+, backports 3.50.7 / 3.44.6), refuse to enable WAL on fresh / non-WAL databases; an already-WAL DB keeps WAL with a warning. - This gate is deliberately RETAINED: an attempt to revert it (theory: DELETE - was "the mode that corrupts") was confounded — its clean WAL result came - from SQLite 3.53.1, which also carries 3.51.0's close()-broken-POSIX-lock - defenses. Re-measured on the bundled 3.50.4 with the lock fix, WAL and - DELETE are both clean, so there is no evidence WAL is safer; keep new - databases out of WAL until a fixed runtime ships. + This gate is deliberately RETAINED: an attempt to revert it was confounded + by a newer SQLite; re-measured on the bundled 3.50.4, WAL and DELETE are + both clean, so there is no evidence WAL is safer. - Invariant on every path (NFS and WAL-reset alike): never downgrade to - DELETE if the on-disk header reports WAL or the mode cannot be read (see - _on_disk_journal_mode). Other gateway/cron/worker connections may hold the - DB open, and a live downgrade destroys their committed-but-uncheckpointed - transactions. + Invariant on every path: never downgrade to DELETE if the on-disk header + reports WAL or the mode cannot be read (see _on_disk_journal_mode). Other + gateway/cron/worker connections may hold the DB open, and a live downgrade + destroys their committed-but-uncheckpointed transactions. - The ERROR is deduplicated per ``db_label``: once per process per DB, so - state.db and kanban.db on one NFS mount each log once. + The ERROR is deduplicated per ``db_label``: once per process per DB. """ from hermes_state import is_sqlite_wal_reset_vulnerable, resolve_journal_mode configured = resolve_journal_mode() @@ -285,9 +283,7 @@ def apply_wal_with_fallback( # accepted DELETE rather than silently returning MEMORY or another mode. if is_sqlite_wal_reset_vulnerable(): return _apply_delete_for_wal_reset_bug( - conn, - db_label=db_label, - require_delete=configured == "delete", + conn, db_label=db_label, require_delete=configured == "delete" ) # Read-only probe — no flock, no checkpoint, no WAL/SHM unlink — so @@ -297,9 +293,7 @@ def apply_wal_with_fallback( if configured == "delete": # Never-live-downgrade keeps WAL; tell the operator their delete did not apply. _log_configured_delete_overridden_once(db_label) - _apply_wal_size_limit(conn) - _apply_macos_checkpoint_barrier(conn) - _enforce_macos_synchronous_full(conn) + _apply_wal_companions(conn) return "wal" # Honor the canonical database.journal_mode setting (on-disk WAL DBs were @@ -307,16 +301,9 @@ def apply_wal_with_fallback( if configured == "delete": if current_mode is None: # Probe failed (locked/busy): another process may hold this DB open - # in WAL, so ownership is not provably exclusive and flipping modes - # could destroy a concurrent writer's committed-but-uncheckpointed - # transactions. Fail loudly — the operator asked for DELETE and we - # cannot verify it. - raise sqlite3.OperationalError( - "could not verify journal mode before applying configured " - "journal_mode=delete (database is locked — possible " - "concurrent openers); refusing to downgrade a database " - "this process does not exclusively own" - ) + # in WAL, so ownership is not provably exclusive. Fail loudly — the + # operator asked for DELETE and we cannot verify it. + raise sqlite3.OperationalError(_CANNOT_VERIFY_DELETE_MSG) actual = _set_journal_mode_no_wait(conn, "DELETE") if actual != "delete": raise sqlite3.OperationalError( @@ -326,34 +313,28 @@ def apply_wal_with_fallback( # Decide BEFORE the flip whether it would overwrite a mode somebody chose: # the probe and page_count are only readable while the file is untouched. - # A 0-page DB has no prior choice, and every caller reaches this before - # creating schema, so brand-new databases stay quiet. + # A 0-page DB has no prior choice, so brand-new databases stay quiet. _upgrading_existing_db = ( - current_mode is not None - and current_mode != "wal" - and _database_has_content(conn) + current_mode is not None and current_mode != "wal" and _database_has_content(conn) ) + def _wal_activated() -> str: + if _upgrading_existing_db: + _log_journal_mode_upgrade_once(db_label, current_mode) + _apply_wal_companions(conn) + return "wal" + try: # ``PRAGMA journal_mode=WAL`` RETURNS the resulting mode. Filesystems # that refuse by *raising* SQLITE_PROTOCOL hit the except branch, but # macOS NFS, SMB/CIFS and the AgentFS NFS overlay refuse WITHOUT raising # and just return the still-effective mode. Trust the row, not the - # absence of an exception, or we report a false "wal", skip the - # fallback ERROR, and leave the DB silently in DELETE. - row = conn.execute("PRAGMA journal_mode=WAL").fetchone() - mode = str(row[0]).strip().lower() if row and row[0] is not None else "" + # absence of an exception. + mode = _mode_from_row(conn.execute("PRAGMA journal_mode=WAL").fetchone()) if mode == "wal": - if _upgrading_existing_db: - _log_journal_mode_upgrade_once(db_label, current_mode) - _apply_wal_size_limit(conn) - _apply_macos_checkpoint_barrier(conn) - _enforce_macos_synchronous_full(conn) - return "wal" + return _wal_activated() # Silent refusal: WAL was not honored, but nothing raised. - silent_exc = WalUnsupportedError( - f"journal_mode=WAL refused without raising (still {mode!r})" - ) + silent_exc = WalUnsupportedError(f"journal_mode=WAL refused without raising (still {mode!r})") if require_wal: raise silent_exc _log_wal_fallback_once(db_label, silent_exc) @@ -365,14 +346,11 @@ def apply_wal_with_fallback( raise msg = str(exc).lower() if not any(marker in msg for marker in _WAL_INCOMPAT_MARKERS): - # Unrelated OperationalError — don't silently swallow. - raise + raise # unrelated OperationalError — don't silently swallow # ``disk i/o error`` is ambiguous: deterministic WAL-incompatibility on - # ZFS / APFS-CoW (SHM corruption under connection bursts), or a one-shot - # transient EIO (page-cache pressure, brief lock contention). Treating - # a transient EIO as a permanent downgrade signal produced mixed-mode - # corruption (process A downgrades to DELETE while siblings set WAL), - # so retry the pragma: transient EIO clears and we return "wal"; + # ZFS / APFS-CoW, or a one-shot transient EIO. Treating a transient EIO + # as a permanent downgrade signal produced mixed-mode corruption, so + # retry the pragma: transient EIO clears and we return "wal"; # deterministic cases keep failing into the guarded DELETE fallback. if "disk i/o error" in msg: for _ in range(2): @@ -384,20 +362,8 @@ def apply_wal_with_fallback( raise exc = retry_exc continue - mode = ( - str(row[0]).strip().lower() - if row and row[0] is not None - else "" - ) - if mode == "wal": - # Transient EIO cleared and the switch went through; same - # header rewrite, so same upgrade signal. - if _upgrading_existing_db: - _log_journal_mode_upgrade_once(db_label, current_mode) - _apply_wal_size_limit(conn) - _apply_macos_checkpoint_barrier(conn) - _enforce_macos_synchronous_full(conn) - return "wal" + if _mode_from_row(row) == "wal": + return _wal_activated() break # Don't downgrade if another process already set WAL on disk, or if the # mode cannot be read (probe blocked by a concurrent opener's locks) — @@ -418,10 +384,9 @@ def _set_journal_mode_no_wait(conn: sqlite3.Connection, mode: str) -> str: The ONLY place a journal-mode switch may be issued for a non-WAL target. Forces ``busy_timeout=0`` so SQLite's exclusivity requirement becomes a concurrent-opener detector: leaving WAL needs exclusive access, so if ANY - other connection (this process or another) holds the DB the pragma fails - immediately with ``database is locked`` instead of waiting out a busy - timeout and sneaking the flip between a concurrent writer's transactions — - exactly how committed-but-uncheckpointed WAL transactions get destroyed. + other connection holds the DB the pragma fails immediately with ``database + is locked`` instead of sneaking the flip between a concurrent writer's + transactions (how committed-but-uncheckpointed WAL transactions die). Callers must treat a raised ``OperationalError`` as "not exclusively owned: leave the journal mode alone", never as retryable. Returns SQLite's @@ -436,8 +401,7 @@ def _set_journal_mode_no_wait(conn: sqlite3.Connection, mode: str) -> str: previous_timeout = 0 conn.execute("PRAGMA busy_timeout=0") try: - row = conn.execute(f"PRAGMA journal_mode={mode}").fetchone() - return str(row[0]).strip().lower() if row and row[0] is not None else "" + return _mode_from_row(conn.execute(f"PRAGMA journal_mode={mode}").fetchone()) finally: try: conn.execute(f"PRAGMA busy_timeout={previous_timeout}") @@ -446,24 +410,19 @@ def _set_journal_mode_no_wait(conn: sqlite3.Connection, mode: str) -> str: def _apply_delete_for_wal_reset_bug( - conn: sqlite3.Connection, - *, - db_label: str, - require_delete: bool = False, + conn: sqlite3.Connection, *, db_label: str, require_delete: bool = False ) -> str: """Avoid enabling WAL when the linked SQLite has the WAL-reset bug. - Already-WAL on disk: leave WAL alone (no live downgrade) and warn. - Mode unreadable (probe blocked by a concurrent opener's locks): not provably exclusive — leave the mode alone and warn. Never treat "could - not read the mode" as "not WAL": that confusion once flipped a live WAL - state.db to DELETE under a concurrent writer, destroying its - committed-but-uncheckpointed transactions. + not read the mode" as "not WAL": that once flipped a live WAL state.db to + DELETE under a concurrent writer, destroying its uncheckpointed commits. - Otherwise: set DELETE (refusing to wait out concurrent openers) and warn. - For an explicit operator request, verify SQLite accepted DELETE. """ current = _on_disk_journal_mode(conn) - if current == "wal": _log_wal_reset_bug_once(db_label, kept_wal=True) if require_delete: @@ -472,24 +431,15 @@ def _apply_delete_for_wal_reset_bug( _log_configured_delete_overridden_once(db_label) # No TRUNCATE / journal_mode=DELETE while other processes may still # hold this WAL DB open; same safety rule as the NFS path. - _apply_wal_size_limit(conn) - _apply_macos_checkpoint_barrier(conn) - _enforce_macos_synchronous_full(conn) + _apply_wal_companions(conn) return "wal" - if current is None: # Probe failed — likely another opener's locks, and the DB may be in # WAL under a live writer. Never flip a mode we cannot even read. if require_delete: - raise sqlite3.OperationalError( - "could not verify journal mode before applying configured " - "journal_mode=delete (database is locked — possible " - "concurrent openers); refusing to downgrade a database " - "this process does not exclusively own" - ) + raise sqlite3.OperationalError(_CANNOT_VERIFY_DELETE_MSG) _log_wal_reset_bug_once(db_label, kept_wal=True, indeterminate=True) return "wal" - actual = "" try: actual = _set_journal_mode_no_wait(conn, "DELETE") @@ -528,8 +478,7 @@ def _wal_reset_repair_hint() -> str: return f"Hermes-managed installs can repair the embedded runtime with `{cmd}`" if method == "docker": return f"update the container image with `{cmd}`" - # nix/nixos - return cmd + return cmd # nix/nixos except Exception: pass return ( @@ -538,25 +487,10 @@ def _wal_reset_repair_hint() -> str: ) -# Dedup state for _log_journal_mode_upgrade_once. -_journal_upgrade_warned_paths: set = set() - - -_journal_upgrade_warned_lock = threading.Lock() - - -def _log_wal_reset_bug_once( - db_label: str, - *, - kept_wal: bool, - indeterminate: bool = False, -) -> None: +def _log_wal_reset_bug_once(db_label: str, *, kept_wal: bool, indeterminate: bool = False) -> None: """Log once per (process, db_label) about the WAL-reset vulnerability path.""" - from hermes_state import _wal_reset_bug_warned_paths - with _wal_reset_bug_warned_lock: - if db_label in _wal_reset_bug_warned_paths: - return - _wal_reset_bug_warned_paths.add(db_label) + if not _warn_once(_wal_reset_bug_warned_lock, "_wal_reset_bug_warned_paths", db_label): + return if indeterminate: action = ( "journal mode could not be verified or exclusively switched " @@ -573,18 +507,13 @@ def _log_wal_reset_bug_once( action = "using journal_mode=DELETE instead of enabling WAL" # Install-type-aware so the warning never promises a repair path that # doesn't exist for git/pip/system Python installs. - repair_hint = _wal_reset_repair_hint() logger.warning( "%s: linked SQLite %s (interpreter %s) is vulnerable to the WAL-reset " "corruption bug (https://sqlite.org/wal.html#walresetbug) — %s. " "Upgrade to SQLite 3.51.3+ (or backports 3.50.7 / 3.44.6); " "%s. See `hermes doctor`. This warning fires once per " "process per database.", - db_label, - sqlite3.sqlite_version, - sys.executable, - action, - repair_hint, + db_label, sqlite3.sqlite_version, sys.executable, action, _wal_reset_repair_hint(), ) @@ -594,20 +523,12 @@ def _log_journal_mode_upgrade_once(db_label: str, previous_mode: str) -> None: ``PRAGMA journal_mode`` is a property of the FILE: switching an existing DB to WAL rewrites its header and outlives the process. Operators do set DELETE on the file directly (the documented WAL-reset-bug mitigation), and - nothing told them the next open would silently put WAL back. - - WARNING, not ERROR: the reverse move is ERROR in ``_log_wal_fallback_once`` - because dropping to DELETE loses concurrency, whereas this direction is - normally desirable (managed_uv repairs DELETE-stuck DBs on update). The - only problem was invisibility, so this names the durable setting without - claiming a degradation. Deduped per process per ``db_label`` because - kanban opens a fresh connection per operation. + nothing told them the next open would silently put WAL back. WARNING, not + ERROR: this direction is normally desirable; only its invisibility was the + problem, so this names the durable setting without claiming a degradation. """ - from hermes_state import _journal_upgrade_warned_paths - with _journal_upgrade_warned_lock: - if db_label in _journal_upgrade_warned_paths: - return - _journal_upgrade_warned_paths.add(db_label) + if not _warn_once(_journal_upgrade_warned_lock, "_journal_upgrade_warned_paths", db_label): + return logger.warning( "%s: on-disk journal_mode was %s and has been switched to WAL. This " "rewrites the database header and persists after this process exits. " @@ -616,9 +537,7 @@ def _log_journal_mode_upgrade_once(db_label: str, previous_mode: str) -> None: "PRAGMA on the file will not survive -- every open re-applies the " "configured mode. Set `database.journal_mode: delete` in config.yaml " "to make it stick. This message fires once per process per database.", - db_label, - previous_mode, - previous_mode, + db_label, previous_mode, previous_mode, ) @@ -627,21 +546,16 @@ def _log_wal_fallback_once(db_label: str, exc: Exception) -> None: ERROR, not WARNING: silently dropping to DELETE is a real concurrency loss (under kanban dispatcher + workers a write blocks readers as SQLITE_BUSY). - Deduped because kanban opens a fresh connection per operation. """ - from hermes_state import _wal_fallback_warned_paths - with _wal_fallback_warned_lock: - if db_label in _wal_fallback_warned_paths: - return - _wal_fallback_warned_paths.add(db_label) + if not _warn_once(_wal_fallback_warned_lock, "_wal_fallback_warned_paths", db_label): + return logger.error( "%s: WAL journal_mode unsupported on this filesystem (%s) — " "falling back to journal_mode=DELETE (slower rollback-journal " "mode; reduces concurrency but works on NFS/SMB/FUSE/ZFS). See " "https://www.sqlite.org/wal.html for details. This message " "fires once per process per database.", - db_label, - exc, + db_label, exc, ) @@ -649,16 +563,12 @@ def _log_configured_delete_overridden_once(db_label: str) -> None: """Log a single ERROR per (process, db_label) when the operator configured ``journal_mode=delete`` but the on-disk DB is already WAL. - Never-live-downgrade keeps WAL (a live downgrade causes mixed-mode - corruption); without this the operator would never learn that - ``database.journal_mode: delete`` had no effect and that a one-time + Never-live-downgrade keeps WAL; without this the operator would never learn + that ``database.journal_mode: delete`` had no effect and that a one-time offline ``PRAGMA journal_mode=DELETE`` (no open connections) is required. """ - from hermes_state import _delete_overridden_warned_paths - with _delete_overridden_warned_lock: - if db_label in _delete_overridden_warned_paths: - return - _delete_overridden_warned_paths.add(db_label) + if not _warn_once(_delete_overridden_warned_lock, "_delete_overridden_warned_paths", db_label): + return logger.error( "%s: database.journal_mode=delete is configured but the on-disk " "database is already WAL; keeping WAL (a live downgrade under open " @@ -675,17 +585,8 @@ def _log_configured_delete_overridden_once(db_label: str) -> None: # --------------------------------------------------------------------------- # Operators write synchronous as a name; mapped here rather than passed through # so a typo becomes a warning instead of a silently different durability level. -_SYNCHRONOUS_LEVELS: Dict[str, int] = { - "OFF": 0, - "NORMAL": 1, - "FULL": 2, - "EXTRA": 3, -} - - +_SYNCHRONOUS_LEVELS: Dict[str, int] = {"OFF": 0, "NORMAL": 1, "FULL": 2, "EXTRA": 3} _SYNCHRONOUS_NAMES: Dict[int, str] = {v: k for k, v in _SYNCHRONOUS_LEVELS.items()} - - _SYNCHRONOUS_FULL = 2 @@ -715,30 +616,23 @@ def resolve_synchronous_level(raw_value: Any) -> Optional[int]: return value if value in _SYNCHRONOUS_NAMES else None -def _apply_synchronous_pragma( - conn: sqlite3.Connection, - raw_value: Any, - *, - db_label: str, -) -> None: +def _apply_synchronous_pragma(conn: sqlite3.Connection, raw_value: Any, *, db_label: str) -> None: """Set ``PRAGMA synchronous`` from config, never below FULL on macOS. Kept out of the integer loop in :func:`apply_database_pragmas`: this PRAGMA decides whether a commit is on the platter, so an unrecognised value must not fall through to "SQLite default" the way a bad ``cache_size`` can. - - Darwin floor: :func:`_enforce_macos_synchronous_full` runs during - ``apply_wal_with_fallback()`` and this runs after it, so a configured - ``NORMAL`` would otherwise silently undo the macOS btree protection. - Raising the level on macOS is allowed; lowering it is refused out loud. + Darwin floor: :func:`_enforce_macos_synchronous_full` runs during WAL + activation and this runs after it, so a configured ``NORMAL`` would + otherwise silently undo the macOS btree protection. Raising the level on + macOS is allowed; lowering it is refused out loud. """ level = resolve_synchronous_level(raw_value) if level is None: logger.warning( "%s: ignoring unrecognized database.synchronous=%r " "(expected OFF, NORMAL, FULL, EXTRA, or 0-3)", - db_label, - raw_value, + db_label, raw_value, ) return if sys.platform == "darwin" and level < _SYNCHRONOUS_FULL: @@ -747,8 +641,7 @@ def _apply_synchronous_pragma( "Darwin's fsync() does not guarantee write ordering, so a lower " "level readmits the half-written btree pages FULL exists to " "prevent.", - db_label, - _SYNCHRONOUS_NAMES[level], + db_label, _SYNCHRONOUS_NAMES[level], ) return try: @@ -757,27 +650,22 @@ def _apply_synchronous_pragma( pass -def apply_database_pragmas( - conn: sqlite3.Connection, - *, - db_label: str = "state.db", -) -> None: +def apply_database_pragmas(conn: sqlite3.Connection, *, db_label: str = "state.db") -> None: """Apply optional performance and WAL-sizing PRAGMAs from ``config.yaml``. Journal mode is NOT handled here — ``database.journal_mode`` is owned by - :func:`resolve_journal_mode` inside :func:`apply_wal_with_fallback`, under - all the safety guards. + :func:`resolve_journal_mode` inside :func:`apply_wal_with_fallback`. Keys under ``database:``: ``cache_size`` (negative = KiB, positive = pages), ``mmap_size`` (bytes, 0 = disabled), ``temp_store`` (0-3), ``wal_autocheckpoint`` (pages), ``journal_size_limit`` (bytes), and ``synchronous`` (``OFF``/``NORMAL``/``FULL``/``EXTRA`` or ``0``-``3``). - Unset ``synchronous`` leaves SQLite's default, a *compile-time* constant - (``SQLITE_DEFAULT_WAL_SYNCHRONOUS``) that differs between bundled, distro - and Homebrew builds; setting it explicitly is the only way to know. + Unset ``synchronous`` leaves SQLite's compile-time default, which differs + between bundled, distro and Homebrew builds. Best-effort: config load or pragma failures are ignored so DB init never - breaks on a malformed ``database:`` section. + breaks on a malformed ``database:`` section. Applied to ALL connection + types: writer, read_only, WAL per-thread readers. """ try: # Local import avoids a circular import with hermes_cli.config. @@ -786,36 +674,21 @@ def apply_database_pragmas( cfg = load_config_readonly() except Exception: return - - # Applied to ALL connection types: writer, read_only, WAL per-thread readers. - for pragma_name in ( - "cache_size", - "mmap_size", - "temp_store", - "wal_autocheckpoint", - "journal_size_limit", - ): + for pragma_name in ("cache_size", "mmap_size", "temp_store", "wal_autocheckpoint", "journal_size_limit"): raw_value = cfg_get(cfg, "database", pragma_name, default=None) if raw_value is None: continue try: value = int(str(raw_value).strip()) except (TypeError, ValueError): - logger.warning( - "%s: ignoring non-integer database.%s=%r", - db_label, - pragma_name, - raw_value, - ) + logger.warning("%s: ignoring non-integer database.%s=%r", db_label, pragma_name, raw_value) continue try: conn.execute(f"PRAGMA {pragma_name}={value}") except sqlite3.OperationalError: pass - # Last: the sizing pragmas above cannot change durability, and the macOS - # enforcement ran earlier during WAL activation (see _apply_synchronous_pragma - # for why that ordering needs an explicit floor rather than an override). + # enforcement ran earlier during WAL activation (see _apply_synchronous_pragma). raw_synchronous = cfg_get(cfg, "database", "synchronous", default=None) if raw_synchronous is not None: _apply_synchronous_pragma(conn, raw_synchronous, db_label=db_label)