refactor(state): wal — _warn_once/_mode_from_row/_apply_wal_companions helpers, shared delete-verify message, compact docs

This commit is contained in:
Teknium
2026-09-02 16:02:42 -07:00
parent ffef90ae7d
commit 176458cdd9

View File

@@ -30,42 +30,56 @@ logger = logging.getLogger("hermes_state")
# filesystems (NFS, SMB/CIFS, some FUSE, WSL1) don't provide reliably — there
# ``PRAGMA journal_mode=WAL`` raises ``locking protocol`` (SQLITE_PROTOCOL).
# ZFS instead corrupts the -shm file under concurrent connection bursts (COW +
# mmap), presenting as ``disk I/O error``. Propagating either would silently
# break everything backed by state.db/kanban.db, so we fall back to
# ``journal_mode=DELETE`` (pre-WAL default, works on NFS/ZFS): readers block
# during a write, but it works. The WAL-reset-bug gate and the
# never-live-downgrade invariant are documented on apply_wal_with_fallback.
# mmap), presenting as ``disk I/O error``. Either would silently break
# everything backed by state.db/kanban.db, so we fall back to
# ``journal_mode=DELETE`` (works on NFS/ZFS; readers block during a write).
_WAL_INCOMPAT_MARKERS = (
"locking protocol", # SQLITE_PROTOCOL on NFS/SMB
"not authorized", # Some FUSE mounts block WAL pragma outright
"disk i/o error", # ZFS SHM corruption under concurrent connections
)
# SQLite's default is -1 (unlimited), so state.db-wal would keep the high-water
# mark of the largest-ever transaction forever. See _apply_wal_size_limit().
_WAL_SIZE_LIMIT_BYTES = 64 * 1024 * 1024 # 64 MiB
# Dedup sets: kanban_db.connect() runs on every kanban operation, so an
# undeduped fallback log line would repeat per connection and fill errors.log.
# Once-per-process-per-db_label dedup sets: kanban_db.connect() runs on every
# kanban operation, so an undeduped log line would repeat per connection.
# Tests clear these through ``hermes_state.<name>``; ``_warn_once`` resolves
# the set through hermes_state at call time for the same reason.
_wal_fallback_warned_paths: set[str] = set()
_wal_fallback_warned_lock = threading.Lock()
_wal_reset_bug_warned_paths: set[str] = set()
_wal_reset_bug_warned_lock = threading.Lock()
# "configured delete overridden by on-disk WAL" ERROR.
_delete_overridden_warned_paths: set[str] = set()
_delete_overridden_warned_lock = threading.Lock()
# Dedup state for _log_journal_mode_upgrade_once.
_journal_upgrade_warned_paths: set = set()
_journal_upgrade_warned_lock = threading.Lock()
_CANNOT_VERIFY_DELETE_MSG = (
"could not verify journal mode before applying configured "
"journal_mode=delete (database is locked — possible "
"concurrent openers); refusing to downgrade a database "
"this process does not exclusively own"
)
def _warn_once(lock: threading.Lock, set_name: str, key: str) -> bool:
"""True the first time *key* is seen in ``hermes_state.<set_name>``."""
import hermes_state
seen = getattr(hermes_state, set_name)
with lock:
if key in seen:
return False
seen.add(key)
return True
def _mode_from_row(row) -> str:
"""Lower-cased mode from a ``PRAGMA journal_mode`` row, ``""`` if no row."""
return str(row[0]).strip().lower() if row and row[0] is not None else ""
def _on_disk_journal_mode(conn: sqlite3.Connection) -> Optional[str]:
@@ -96,9 +110,7 @@ def _on_disk_journal_mode(conn: sqlite3.Connection) -> Optional[str]:
return None
return str(mode).strip().lower() if mode is not None else None
if last_exc is not None:
logger.debug(
"_on_disk_journal_mode: retries exhausted on disk read (%s)", last_exc
)
logger.debug("_on_disk_journal_mode: retries exhausted on disk read (%s)", last_exc)
return None
@@ -107,15 +119,10 @@ def _apply_wal_size_limit(conn: sqlite3.Connection) -> None:
SQLite's default ``journal_size_limit`` is -1: a checkpointed WAL is reused
in place, never truncated, so ``state.db-wal`` keeps the high-water mark
of the largest transaction ever run. One bulk op strands gigabytes —
``hermes sessions optimize`` on a 3 GB state.db left a 3 GB WAL and filled
the disk, so the maintenance command was self-defeating on the largest
DBs. With a limit, each checkpoint truncates the WAL back to it; 64 MiB is
above normal transaction sizes (steady-state commits never pay a truncate)
while capping slack predictably. kanban_db uses ``wal_autocheckpoint=100``.
Best-effort: never raises — failure only costs disk slack and must not
prevent the database from opening.
of the largest transaction ever run (a 3 GB optimize left a 3 GB WAL).
With a limit, each checkpoint truncates the WAL back to it; 64 MiB is
above normal transaction sizes while capping slack predictably.
Best-effort: failure only costs disk slack and must not prevent opening.
"""
try:
conn.execute(f"PRAGMA journal_size_limit={_WAL_SIZE_LIMIT_BYTES}")
@@ -127,12 +134,11 @@ def _apply_macos_checkpoint_barrier(conn: sqlite3.Connection) -> None:
"""Enable ``PRAGMA checkpoint_fullfsync`` on macOS (no-op elsewhere).
Apple's ``fsync(2)`` guarantees neither data-on-platter nor write ordering,
so WAL's corruption-safety assumption fails on Darwin without ``F_FULLFSYNC``.
A launchd shutdown drops the page cache (power-loss for in-flight pages), so
a checkpoint that "reported" durable can leave a malformed ``state.db``;
a plain in-session kill survives via the page cache. The barrier applies
only at checkpoint boundaries (~+0.1 ms/commit vs ~+4 ms for
``fullfsync=1``). Best-effort: never raises.
so WAL's corruption-safety assumption fails on Darwin without ``F_FULLFSYNC``:
a launchd shutdown drops the page cache and a checkpoint that "reported"
durable can leave a malformed ``state.db``. The barrier applies only at
checkpoint boundaries (~+0.1 ms/commit vs ~+4 ms for ``fullfsync=1``).
Best-effort: never raises.
"""
if sys.platform != "darwin":
return
@@ -145,11 +151,10 @@ def _apply_macos_checkpoint_barrier(conn: sqlite3.Connection) -> None:
def _enforce_macos_synchronous_full(conn: sqlite3.Connection) -> None:
"""Enforce ``PRAGMA synchronous=FULL`` on macOS to prevent btree corruption.
With NORMAL, a WAL checkpoint racing process termination (launchd shutdown)
can leave half-written btree pages (``btreeInitPage error 11``) because
Darwin's ``fsync()`` guarantees neither ordering nor durability. Called
after every successful WAL activation so a prior connection's NORMAL never
sticks. Best-effort: never raises.
With NORMAL, a WAL checkpoint racing process termination can leave
half-written btree pages (``btreeInitPage error 11``). Called after every
successful WAL activation so a prior connection's NORMAL never sticks.
Best-effort: never raises.
"""
if sys.platform != "darwin":
return
@@ -159,9 +164,14 @@ def _enforce_macos_synchronous_full(conn: sqlite3.Connection) -> None:
pass
def is_sqlite_wal_reset_vulnerable(
version_info: Optional[tuple] = None,
) -> bool:
def _apply_wal_companions(conn: sqlite3.Connection) -> None:
"""The settings every WAL activation carries: size limit + macOS barriers."""
_apply_wal_size_limit(conn)
_apply_macos_checkpoint_barrier(conn)
_enforce_macos_synchronous_full(conn)
def is_sqlite_wal_reset_vulnerable(version_info: Optional[tuple] = None) -> bool:
"""True when the linked SQLite has the WAL-reset bug (3.7.0–3.51.2;
fixed 3.51.3+, backports 3.50.7 / 3.44.6). Pre-WAL libraries are safe.
https://sqlite.org/wal.html#walresetbug
@@ -190,8 +200,7 @@ def _database_has_content(conn: sqlite3.Connection) -> bool:
``PRAGMA page_count`` is a lock-free header read. Fail-quiet: any error
answers False, because the only caller gates a warning on this and an
unknown-answer warning would fire on every fresh database — exactly where
there is provably no operator choice being overwritten.
unknown-answer warning would fire on every fresh database.
"""
try:
row = conn.execute("PRAGMA page_count").fetchone()
@@ -222,7 +231,6 @@ def resolve_journal_mode() -> str:
raw = database.get("journal_mode", "wal")
except Exception:
return "wal"
if not isinstance(raw, str):
return "wal"
mode = raw.strip().lower()
@@ -239,43 +247,33 @@ class WalUnsupportedError(sqlite3.OperationalError):
def apply_wal_with_fallback(
conn: sqlite3.Connection,
*,
db_label: str = "state.db",
require_wal: bool = False,
conn: sqlite3.Connection, *, db_label: str = "state.db", require_wal: bool = False
) -> str:
"""Set ``journal_mode=WAL`` on ``conn``, falling back to DELETE on failure.
Returns the mode actually set (``"wal"`` or ``"delete"``). Shared by
:class:`SessionDB` and ``hermes_cli.kanban_db.connect`` for identical
fallback behavior.
:class:`SessionDB` and ``hermes_cli.kanban_db.connect``.
On WAL-incompatible filesystems (NFS, SMB, some FUSE, ZFS) SQLite either
raises ``OperationalError`` ("locking protocol" / "disk I/O error") or —
macOS NFS / SMB / AgentFS NFS overlay — silently refuses and leaves the DB
in DELETE. Either way we log at ERROR (a write now blocks readers — a real
concurrency loss) and fall back to DELETE so the feature keeps working.
``require_wal=True`` raises :class:`WalUnsupportedError` instead; all
current callers keep the default so NFS-homed installs work.
On WAL-incompatible filesystems SQLite either raises ``OperationalError``
("locking protocol" / "disk I/O error") or — macOS NFS / SMB / AgentFS NFS
overlay — silently refuses and leaves the DB in DELETE. Either way we log
at ERROR (a write now blocks readers) and fall back to DELETE so the
feature keeps working. ``require_wal=True`` raises
:class:`WalUnsupportedError` instead.
On SQLite builds with the WAL-reset bug (https://sqlite.org/wal.html#walresetbug,
fixed 3.51.3+, backports 3.50.7 / 3.44.6), refuse to enable WAL on
fresh / non-WAL databases; an already-WAL DB keeps WAL with a warning.
This gate is deliberately RETAINED: an attempt to revert it (theory: DELETE
was "the mode that corrupts") was confounded — its clean WAL result came
from SQLite 3.53.1, which also carries 3.51.0's close()-broken-POSIX-lock
defenses. Re-measured on the bundled 3.50.4 with the lock fix, WAL and
DELETE are both clean, so there is no evidence WAL is safer; keep new
databases out of WAL until a fixed runtime ships.
This gate is deliberately RETAINED: an attempt to revert it was confounded
by a newer SQLite; re-measured on the bundled 3.50.4, WAL and DELETE are
both clean, so there is no evidence WAL is safer.
Invariant on every path (NFS and WAL-reset alike): never downgrade to
DELETE if the on-disk header reports WAL or the mode cannot be read (see
_on_disk_journal_mode). Other gateway/cron/worker connections may hold the
DB open, and a live downgrade destroys their committed-but-uncheckpointed
transactions.
Invariant on every path: never downgrade to DELETE if the on-disk header
reports WAL or the mode cannot be read (see _on_disk_journal_mode). Other
gateway/cron/worker connections may hold the DB open, and a live downgrade
destroys their committed-but-uncheckpointed transactions.
The ERROR is deduplicated per ``db_label``: once per process per DB, so
state.db and kanban.db on one NFS mount each log once.
The ERROR is deduplicated per ``db_label``: once per process per DB.
"""
from hermes_state import is_sqlite_wal_reset_vulnerable, resolve_journal_mode
configured = resolve_journal_mode()
@@ -285,9 +283,7 @@ def apply_wal_with_fallback(
# accepted DELETE rather than silently returning MEMORY or another mode.
if is_sqlite_wal_reset_vulnerable():
return _apply_delete_for_wal_reset_bug(
conn,
db_label=db_label,
require_delete=configured == "delete",
conn, db_label=db_label, require_delete=configured == "delete"
)
# Read-only probe — no flock, no checkpoint, no WAL/SHM unlink — so
@@ -297,9 +293,7 @@ def apply_wal_with_fallback(
if configured == "delete":
# Never-live-downgrade keeps WAL; tell the operator their delete did not apply.
_log_configured_delete_overridden_once(db_label)
_apply_wal_size_limit(conn)
_apply_macos_checkpoint_barrier(conn)
_enforce_macos_synchronous_full(conn)
_apply_wal_companions(conn)
return "wal"
# Honor the canonical database.journal_mode setting (on-disk WAL DBs were
@@ -307,16 +301,9 @@ def apply_wal_with_fallback(
if configured == "delete":
if current_mode is None:
# Probe failed (locked/busy): another process may hold this DB open
# in WAL, so ownership is not provably exclusive and flipping modes
# could destroy a concurrent writer's committed-but-uncheckpointed
# transactions. Fail loudly — the operator asked for DELETE and we
# cannot verify it.
raise sqlite3.OperationalError(
"could not verify journal mode before applying configured "
"journal_mode=delete (database is locked — possible "
"concurrent openers); refusing to downgrade a database "
"this process does not exclusively own"
)
# in WAL, so ownership is not provably exclusive. Fail loudly — the
# operator asked for DELETE and we cannot verify it.
raise sqlite3.OperationalError(_CANNOT_VERIFY_DELETE_MSG)
actual = _set_journal_mode_no_wait(conn, "DELETE")
if actual != "delete":
raise sqlite3.OperationalError(
@@ -326,34 +313,28 @@ def apply_wal_with_fallback(
# Decide BEFORE the flip whether it would overwrite a mode somebody chose:
# the probe and page_count are only readable while the file is untouched.
# A 0-page DB has no prior choice, and every caller reaches this before
# creating schema, so brand-new databases stay quiet.
# A 0-page DB has no prior choice, so brand-new databases stay quiet.
_upgrading_existing_db = (
current_mode is not None
and current_mode != "wal"
and _database_has_content(conn)
current_mode is not None and current_mode != "wal" and _database_has_content(conn)
)
def _wal_activated() -> str:
if _upgrading_existing_db:
_log_journal_mode_upgrade_once(db_label, current_mode)
_apply_wal_companions(conn)
return "wal"
try:
# ``PRAGMA journal_mode=WAL`` RETURNS the resulting mode. Filesystems
# that refuse by *raising* SQLITE_PROTOCOL hit the except branch, but
# macOS NFS, SMB/CIFS and the AgentFS NFS overlay refuse WITHOUT raising
# and just return the still-effective mode. Trust the row, not the
# absence of an exception, or we report a false "wal", skip the
# fallback ERROR, and leave the DB silently in DELETE.
row = conn.execute("PRAGMA journal_mode=WAL").fetchone()
mode = str(row[0]).strip().lower() if row and row[0] is not None else ""
# absence of an exception.
mode = _mode_from_row(conn.execute("PRAGMA journal_mode=WAL").fetchone())
if mode == "wal":
if _upgrading_existing_db:
_log_journal_mode_upgrade_once(db_label, current_mode)
_apply_wal_size_limit(conn)
_apply_macos_checkpoint_barrier(conn)
_enforce_macos_synchronous_full(conn)
return "wal"
return _wal_activated()
# Silent refusal: WAL was not honored, but nothing raised.
silent_exc = WalUnsupportedError(
f"journal_mode=WAL refused without raising (still {mode!r})"
)
silent_exc = WalUnsupportedError(f"journal_mode=WAL refused without raising (still {mode!r})")
if require_wal:
raise silent_exc
_log_wal_fallback_once(db_label, silent_exc)
@@ -365,14 +346,11 @@ def apply_wal_with_fallback(
raise
msg = str(exc).lower()
if not any(marker in msg for marker in _WAL_INCOMPAT_MARKERS):
# Unrelated OperationalError — don't silently swallow.
raise
raise # unrelated OperationalError — don't silently swallow
# ``disk i/o error`` is ambiguous: deterministic WAL-incompatibility on
# ZFS / APFS-CoW (SHM corruption under connection bursts), or a one-shot
# transient EIO (page-cache pressure, brief lock contention). Treating
# a transient EIO as a permanent downgrade signal produced mixed-mode
# corruption (process A downgrades to DELETE while siblings set WAL),
# so retry the pragma: transient EIO clears and we return "wal";
# ZFS / APFS-CoW, or a one-shot transient EIO. Treating a transient EIO
# as a permanent downgrade signal produced mixed-mode corruption, so
# retry the pragma: transient EIO clears and we return "wal";
# deterministic cases keep failing into the guarded DELETE fallback.
if "disk i/o error" in msg:
for _ in range(2):
@@ -384,20 +362,8 @@ def apply_wal_with_fallback(
raise
exc = retry_exc
continue
mode = (
str(row[0]).strip().lower()
if row and row[0] is not None
else ""
)
if mode == "wal":
# Transient EIO cleared and the switch went through; same
# header rewrite, so same upgrade signal.
if _upgrading_existing_db:
_log_journal_mode_upgrade_once(db_label, current_mode)
_apply_wal_size_limit(conn)
_apply_macos_checkpoint_barrier(conn)
_enforce_macos_synchronous_full(conn)
return "wal"
if _mode_from_row(row) == "wal":
return _wal_activated()
break
# Don't downgrade if another process already set WAL on disk, or if the
# mode cannot be read (probe blocked by a concurrent opener's locks) —
@@ -418,10 +384,9 @@ def _set_journal_mode_no_wait(conn: sqlite3.Connection, mode: str) -> str:
The ONLY place a journal-mode switch may be issued for a non-WAL target.
Forces ``busy_timeout=0`` so SQLite's exclusivity requirement becomes a
concurrent-opener detector: leaving WAL needs exclusive access, so if ANY
other connection (this process or another) holds the DB the pragma fails
immediately with ``database is locked`` instead of waiting out a busy
timeout and sneaking the flip between a concurrent writer's transactions —
exactly how committed-but-uncheckpointed WAL transactions get destroyed.
other connection holds the DB the pragma fails immediately with ``database
is locked`` instead of sneaking the flip between a concurrent writer's
transactions (how committed-but-uncheckpointed WAL transactions die).
Callers must treat a raised ``OperationalError`` as "not exclusively
owned: leave the journal mode alone", never as retryable. Returns SQLite's
@@ -436,8 +401,7 @@ def _set_journal_mode_no_wait(conn: sqlite3.Connection, mode: str) -> str:
previous_timeout = 0
conn.execute("PRAGMA busy_timeout=0")
try:
row = conn.execute(f"PRAGMA journal_mode={mode}").fetchone()
return str(row[0]).strip().lower() if row and row[0] is not None else ""
return _mode_from_row(conn.execute(f"PRAGMA journal_mode={mode}").fetchone())
finally:
try:
conn.execute(f"PRAGMA busy_timeout={previous_timeout}")
@@ -446,24 +410,19 @@ def _set_journal_mode_no_wait(conn: sqlite3.Connection, mode: str) -> str:
def _apply_delete_for_wal_reset_bug(
conn: sqlite3.Connection,
*,
db_label: str,
require_delete: bool = False,
conn: sqlite3.Connection, *, db_label: str, require_delete: bool = False
) -> str:
"""Avoid enabling WAL when the linked SQLite has the WAL-reset bug.
- Already-WAL on disk: leave WAL alone (no live downgrade) and warn.
- Mode unreadable (probe blocked by a concurrent opener's locks): not
provably exclusive — leave the mode alone and warn. Never treat "could
not read the mode" as "not WAL": that confusion once flipped a live WAL
state.db to DELETE under a concurrent writer, destroying its
committed-but-uncheckpointed transactions.
not read the mode" as "not WAL": that once flipped a live WAL state.db to
DELETE under a concurrent writer, destroying its uncheckpointed commits.
- Otherwise: set DELETE (refusing to wait out concurrent openers) and warn.
- For an explicit operator request, verify SQLite accepted DELETE.
"""
current = _on_disk_journal_mode(conn)
if current == "wal":
_log_wal_reset_bug_once(db_label, kept_wal=True)
if require_delete:
@@ -472,24 +431,15 @@ def _apply_delete_for_wal_reset_bug(
_log_configured_delete_overridden_once(db_label)
# No TRUNCATE / journal_mode=DELETE while other processes may still
# hold this WAL DB open; same safety rule as the NFS path.
_apply_wal_size_limit(conn)
_apply_macos_checkpoint_barrier(conn)
_enforce_macos_synchronous_full(conn)
_apply_wal_companions(conn)
return "wal"
if current is None:
# Probe failed — likely another opener's locks, and the DB may be in
# WAL under a live writer. Never flip a mode we cannot even read.
if require_delete:
raise sqlite3.OperationalError(
"could not verify journal mode before applying configured "
"journal_mode=delete (database is locked — possible "
"concurrent openers); refusing to downgrade a database "
"this process does not exclusively own"
)
raise sqlite3.OperationalError(_CANNOT_VERIFY_DELETE_MSG)
_log_wal_reset_bug_once(db_label, kept_wal=True, indeterminate=True)
return "wal"
actual = ""
try:
actual = _set_journal_mode_no_wait(conn, "DELETE")
@@ -528,8 +478,7 @@ def _wal_reset_repair_hint() -> str:
return f"Hermes-managed installs can repair the embedded runtime with `{cmd}`"
if method == "docker":
return f"update the container image with `{cmd}`"
# nix/nixos
return cmd
return cmd # nix/nixos
except Exception:
pass
return (
@@ -538,25 +487,10 @@ def _wal_reset_repair_hint() -> str:
)
# Dedup state for _log_journal_mode_upgrade_once.
_journal_upgrade_warned_paths: set = set()
_journal_upgrade_warned_lock = threading.Lock()
def _log_wal_reset_bug_once(
db_label: str,
*,
kept_wal: bool,
indeterminate: bool = False,
) -> None:
def _log_wal_reset_bug_once(db_label: str, *, kept_wal: bool, indeterminate: bool = False) -> None:
"""Log once per (process, db_label) about the WAL-reset vulnerability path."""
from hermes_state import _wal_reset_bug_warned_paths
with _wal_reset_bug_warned_lock:
if db_label in _wal_reset_bug_warned_paths:
return
_wal_reset_bug_warned_paths.add(db_label)
if not _warn_once(_wal_reset_bug_warned_lock, "_wal_reset_bug_warned_paths", db_label):
return
if indeterminate:
action = (
"journal mode could not be verified or exclusively switched "
@@ -573,18 +507,13 @@ def _log_wal_reset_bug_once(
action = "using journal_mode=DELETE instead of enabling WAL"
# Install-type-aware so the warning never promises a repair path that
# doesn't exist for git/pip/system Python installs.
repair_hint = _wal_reset_repair_hint()
logger.warning(
"%s: linked SQLite %s (interpreter %s) is vulnerable to the WAL-reset "
"corruption bug (https://sqlite.org/wal.html#walresetbug) — %s. "
"Upgrade to SQLite 3.51.3+ (or backports 3.50.7 / 3.44.6); "
"%s. See `hermes doctor`. This warning fires once per "
"process per database.",
db_label,
sqlite3.sqlite_version,
sys.executable,
action,
repair_hint,
db_label, sqlite3.sqlite_version, sys.executable, action, _wal_reset_repair_hint(),
)
@@ -594,20 +523,12 @@ def _log_journal_mode_upgrade_once(db_label: str, previous_mode: str) -> None:
``PRAGMA journal_mode`` is a property of the FILE: switching an existing DB
to WAL rewrites its header and outlives the process. Operators do set
DELETE on the file directly (the documented WAL-reset-bug mitigation), and
nothing told them the next open would silently put WAL back.
WARNING, not ERROR: the reverse move is ERROR in ``_log_wal_fallback_once``
because dropping to DELETE loses concurrency, whereas this direction is
normally desirable (managed_uv repairs DELETE-stuck DBs on update). The
only problem was invisibility, so this names the durable setting without
claiming a degradation. Deduped per process per ``db_label`` because
kanban opens a fresh connection per operation.
nothing told them the next open would silently put WAL back. WARNING, not
ERROR: this direction is normally desirable; only its invisibility was the
problem, so this names the durable setting without claiming a degradation.
"""
from hermes_state import _journal_upgrade_warned_paths
with _journal_upgrade_warned_lock:
if db_label in _journal_upgrade_warned_paths:
return
_journal_upgrade_warned_paths.add(db_label)
if not _warn_once(_journal_upgrade_warned_lock, "_journal_upgrade_warned_paths", db_label):
return
logger.warning(
"%s: on-disk journal_mode was %s and has been switched to WAL. This "
"rewrites the database header and persists after this process exits. "
@@ -616,9 +537,7 @@ def _log_journal_mode_upgrade_once(db_label: str, previous_mode: str) -> None:
"PRAGMA on the file will not survive -- every open re-applies the "
"configured mode. Set `database.journal_mode: delete` in config.yaml "
"to make it stick. This message fires once per process per database.",
db_label,
previous_mode,
previous_mode,
db_label, previous_mode, previous_mode,
)
@@ -627,21 +546,16 @@ def _log_wal_fallback_once(db_label: str, exc: Exception) -> None:
ERROR, not WARNING: silently dropping to DELETE is a real concurrency loss
(under kanban dispatcher + workers a write blocks readers as SQLITE_BUSY).
Deduped because kanban opens a fresh connection per operation.
"""
from hermes_state import _wal_fallback_warned_paths
with _wal_fallback_warned_lock:
if db_label in _wal_fallback_warned_paths:
return
_wal_fallback_warned_paths.add(db_label)
if not _warn_once(_wal_fallback_warned_lock, "_wal_fallback_warned_paths", db_label):
return
logger.error(
"%s: WAL journal_mode unsupported on this filesystem (%s) — "
"falling back to journal_mode=DELETE (slower rollback-journal "
"mode; reduces concurrency but works on NFS/SMB/FUSE/ZFS). See "
"https://www.sqlite.org/wal.html for details. This message "
"fires once per process per database.",
db_label,
exc,
db_label, exc,
)
@@ -649,16 +563,12 @@ def _log_configured_delete_overridden_once(db_label: str) -> None:
"""Log a single ERROR per (process, db_label) when the operator configured
``journal_mode=delete`` but the on-disk DB is already WAL.
Never-live-downgrade keeps WAL (a live downgrade causes mixed-mode
corruption); without this the operator would never learn that
``database.journal_mode: delete`` had no effect and that a one-time
Never-live-downgrade keeps WAL; without this the operator would never learn
that ``database.journal_mode: delete`` had no effect and that a one-time
offline ``PRAGMA journal_mode=DELETE`` (no open connections) is required.
"""
from hermes_state import _delete_overridden_warned_paths
with _delete_overridden_warned_lock:
if db_label in _delete_overridden_warned_paths:
return
_delete_overridden_warned_paths.add(db_label)
if not _warn_once(_delete_overridden_warned_lock, "_delete_overridden_warned_paths", db_label):
return
logger.error(
"%s: database.journal_mode=delete is configured but the on-disk "
"database is already WAL; keeping WAL (a live downgrade under open "
@@ -675,17 +585,8 @@ def _log_configured_delete_overridden_once(db_label: str) -> None:
# ---------------------------------------------------------------------------
# Operators write synchronous as a name; mapped here rather than passed through
# so a typo becomes a warning instead of a silently different durability level.
_SYNCHRONOUS_LEVELS: Dict[str, int] = {
"OFF": 0,
"NORMAL": 1,
"FULL": 2,
"EXTRA": 3,
}
_SYNCHRONOUS_LEVELS: Dict[str, int] = {"OFF": 0, "NORMAL": 1, "FULL": 2, "EXTRA": 3}
_SYNCHRONOUS_NAMES: Dict[int, str] = {v: k for k, v in _SYNCHRONOUS_LEVELS.items()}
_SYNCHRONOUS_FULL = 2
@@ -715,30 +616,23 @@ def resolve_synchronous_level(raw_value: Any) -> Optional[int]:
return value if value in _SYNCHRONOUS_NAMES else None
def _apply_synchronous_pragma(
conn: sqlite3.Connection,
raw_value: Any,
*,
db_label: str,
) -> None:
def _apply_synchronous_pragma(conn: sqlite3.Connection, raw_value: Any, *, db_label: str) -> None:
"""Set ``PRAGMA synchronous`` from config, never below FULL on macOS.
Kept out of the integer loop in :func:`apply_database_pragmas`: this PRAGMA
decides whether a commit is on the platter, so an unrecognised value must
not fall through to "SQLite default" the way a bad ``cache_size`` can.
Darwin floor: :func:`_enforce_macos_synchronous_full` runs during
``apply_wal_with_fallback()`` and this runs after it, so a configured
``NORMAL`` would otherwise silently undo the macOS btree protection.
Raising the level on macOS is allowed; lowering it is refused out loud.
Darwin floor: :func:`_enforce_macos_synchronous_full` runs during WAL
activation and this runs after it, so a configured ``NORMAL`` would
otherwise silently undo the macOS btree protection. Raising the level on
macOS is allowed; lowering it is refused out loud.
"""
level = resolve_synchronous_level(raw_value)
if level is None:
logger.warning(
"%s: ignoring unrecognized database.synchronous=%r "
"(expected OFF, NORMAL, FULL, EXTRA, or 0-3)",
db_label,
raw_value,
db_label, raw_value,
)
return
if sys.platform == "darwin" and level < _SYNCHRONOUS_FULL:
@@ -747,8 +641,7 @@ def _apply_synchronous_pragma(
"Darwin's fsync() does not guarantee write ordering, so a lower "
"level readmits the half-written btree pages FULL exists to "
"prevent.",
db_label,
_SYNCHRONOUS_NAMES[level],
db_label, _SYNCHRONOUS_NAMES[level],
)
return
try:
@@ -757,27 +650,22 @@ def _apply_synchronous_pragma(
pass
def apply_database_pragmas(
conn: sqlite3.Connection,
*,
db_label: str = "state.db",
) -> None:
def apply_database_pragmas(conn: sqlite3.Connection, *, db_label: str = "state.db") -> None:
"""Apply optional performance and WAL-sizing PRAGMAs from ``config.yaml``.
Journal mode is NOT handled here — ``database.journal_mode`` is owned by
:func:`resolve_journal_mode` inside :func:`apply_wal_with_fallback`, under
all the safety guards.
:func:`resolve_journal_mode` inside :func:`apply_wal_with_fallback`.
Keys under ``database:``: ``cache_size`` (negative = KiB, positive =
pages), ``mmap_size`` (bytes, 0 = disabled), ``temp_store`` (0-3),
``wal_autocheckpoint`` (pages), ``journal_size_limit`` (bytes), and
``synchronous`` (``OFF``/``NORMAL``/``FULL``/``EXTRA`` or ``0``-``3``).
Unset ``synchronous`` leaves SQLite's default, a *compile-time* constant
(``SQLITE_DEFAULT_WAL_SYNCHRONOUS``) that differs between bundled, distro
and Homebrew builds; setting it explicitly is the only way to know.
Unset ``synchronous`` leaves SQLite's compile-time default, which differs
between bundled, distro and Homebrew builds.
Best-effort: config load or pragma failures are ignored so DB init never
breaks on a malformed ``database:`` section.
breaks on a malformed ``database:`` section. Applied to ALL connection
types: writer, read_only, WAL per-thread readers.
"""
try:
# Local import avoids a circular import with hermes_cli.config.
@@ -786,36 +674,21 @@ def apply_database_pragmas(
cfg = load_config_readonly()
except Exception:
return
# Applied to ALL connection types: writer, read_only, WAL per-thread readers.
for pragma_name in (
"cache_size",
"mmap_size",
"temp_store",
"wal_autocheckpoint",
"journal_size_limit",
):
for pragma_name in ("cache_size", "mmap_size", "temp_store", "wal_autocheckpoint", "journal_size_limit"):
raw_value = cfg_get(cfg, "database", pragma_name, default=None)
if raw_value is None:
continue
try:
value = int(str(raw_value).strip())
except (TypeError, ValueError):
logger.warning(
"%s: ignoring non-integer database.%s=%r",
db_label,
pragma_name,
raw_value,
)
logger.warning("%s: ignoring non-integer database.%s=%r", db_label, pragma_name, raw_value)
continue
try:
conn.execute(f"PRAGMA {pragma_name}={value}")
except sqlite3.OperationalError:
pass
# Last: the sizing pragmas above cannot change durability, and the macOS
# enforcement ran earlier during WAL activation (see _apply_synchronous_pragma
# for why that ordering needs an explicit floor rather than an override).
# enforcement ran earlier during WAL activation (see _apply_synchronous_pragma).
raw_synchronous = cfg_get(cfg, "database", "synchronous", default=None)
if raw_synchronous is not None:
_apply_synchronous_pragma(conn, raw_synchronous, db_label=db_label)