refactor(state): extract error types/classifiers to hermes_state_errors and the live-DB guard to hermes_state_guard

This commit is contained in:
Teknium
2026-09-02 18:44:39 -07:00
parent 30b4685036
commit e527b0699d
3 changed files with 396 additions and 364 deletions

View File

@@ -45,6 +45,21 @@ from hermes_state_common import ( # noqa: F401 (re-exported; tests import from
FTS_TRIGRAM_SQL, LEGACY_FTS_SQL, LEGACY_FTS_TRIGRAM_SQL, SCHEMA_SQL, SCHEMA_VERSION,
stat_db_file_identity as _stat_db_file_identity,
)
from hermes_state_errors import ( # noqa: F401 (re-exported; the historical import path)
_DB_CORRUPTION_MARKERS, _DELETED_WAL_GENERATION_MSG, _DISK_FULL_MARKERS, _DISK_IO_ERROR_MARKER,
_MALFORMED_DB_MARKERS, _MALFORMED_SCHEMA_MARKERS, _STATE_DB_APPLICATION_ID_OFFSET,
_STATE_DB_CORRUPT_MSG, _STATE_DB_GENERATION_KEY, _STATE_DB_REPLACED_MSG,
_TRANSIENT_SQLITE_MARKERS, PERSISTENCE_ERROR_CAUSES, CompressionSessionBusyError,
CompressionSessionClosedError, DeletedWalGenerationError, SessionCompressionInProgressError,
SessionTurnLeaseLostError, StateDbCorruptError, StateDbReplacedError, _is_no_more_rows,
classify_persistence_error, is_disk_full_error, is_malformed_db_error,
is_malformed_schema_error, is_transient_sqlite_error,
)
from hermes_state_guard import ( # noqa: F401 (re-exported; tests patch hermes_state.<name>)
_PYTEST_LAUNCHER_NAMES, _STATE_DB_GUARD_BYPASS_ENV, _TEST_ISOLATION_MARKER_ENV,
_has_pytest_ancestor, _in_test_context, _is_production_state_db, _process_looks_like_pytest,
_real_platform_state_root, _running_under_pytest, _set_last_init_error, get_last_init_error,
)
from hermes_state_portability import SessionPortabilityMixin
from hermes_state_telegram import SessionTelegramTopicsMixin
from hermes_state_schema import SessionSchemaMixin
@@ -361,6 +376,12 @@ _READ_OPEN_RETRY_SECONDS = 60.0
_READ_ONLY_IOERR_RETRY_ATTEMPTS = 3
_READ_ONLY_IOERR_RETRY_BACKOFF_S = 0.05
def _is_transient_read_only_ioerr(exc: sqlite3.OperationalError, *, attempt: int) -> bool:
"""Retry a read-only open? See _READ_ONLY_IOERR_RETRY_ATTEMPTS: a
persistent IOERR still exhausts the budget and propagates."""
return attempt < _READ_ONLY_IOERR_RETRY_ATTEMPTS and _DISK_IO_ERROR_MARKER in str(exc).lower()
# Ceiling on read-only connections ALIVE at once against one database FILE
# (idle pooled + checked out, summed over every SessionDB on that file). One
# constant for both the pool maxsize and the permit count: a LifoQueue only caps
@@ -583,120 +604,17 @@ def _default_db_path() -> Path:
return get_hermes_home() / "state.db"
# ---------------------------------------------------------------------------
# Live-DB test-isolation guard
# ---------------------------------------------------------------------------
# Field evidence: pytest fixture rows landed in the production state.db and a
# pytest-spawned child flipped the journal mode under the live WAL writer,
# destroying committed transcripts. The hermetic conftest redirects HERMES_HOME
# per test, but any escape (fixture ordering, a child spawned without
# HERMES_HOME, a shell exporting the real home) fell through silently. EVERY
# SessionDB construction resolves its path here, so under pytest a production
# state.db fails hard. Env-based, so subprocess children are protected too.
# Live-DB guard knobs live HERE (not in hermes_state_guard): the hermetic conftest
# monkeypatches ``hermes_state._STATE_DB_GUARD_BYPASS`` / ``_EXTRA_DENY_ROOTS``.
#: Escape hatch for tests that genuinely need the real DB (conftest sets it for
#: ``@pytest.mark.live_system_guard_bypass``); scripts may set it explicitly.
_STATE_DB_GUARD_BYPASS = False
#: Env twin of ``_STATE_DB_GUARD_BYPASS`` for child processes (a module global
#: cannot cross a process boundary, and ancestry arms the guard there).
_STATE_DB_GUARD_BYPASS_ENV = "HERMES_STATE_DB_GUARD_BYPASS"
#: Extra production roots to refuse; conftest injects the pre-sandbox root so
#: custom-HERMES_HOME deployments are covered too.
_STATE_DB_GUARD_EXTRA_DENY_ROOTS: Tuple[Path, ...] = ()
def _real_platform_state_root() -> Optional[Path]:
"""The REAL platform-default Hermes root. Avoids ``Path.home()`` /
``hermes_constants``: tests monkeypatch Path.home to a tempdir while this
module is imported lazily, which would misidentify the hermetic home as
production or miss the real one. ``expanduser`` reads HOME/passwd, which the
conftest never rewrites."""
try:
if sys.platform == "win32":
base = os.environ.get("LOCALAPPDATA", "").strip()
root = (
Path(base) / "hermes"
if base
else Path(os.path.expanduser("~")) / "AppData" / "Local" / "hermes"
)
else:
root = Path(os.path.expanduser("~")) / ".hermes"
return root.resolve()
except Exception:
return None
#: Exported by the hermetic conftest alongside the HERMES_HOME redirect (value:
#: the isolation root). Unlike PYTEST_* (scrubbed by tests that rebuild a child
#: env) it is OURS and inherits by default, so a child carrying it that resolves
#: a production DB is by definition an isolation escape.
_TEST_ISOLATION_MARKER_ENV = "HERMES_TEST_ISOLATION"
def _running_under_pytest() -> bool:
"""True when this process (or a parent test process) is a pytest run."""
return bool(
os.environ.get("PYTEST_CURRENT_TEST")
or os.environ.get("PYTEST_VERSION")
or os.environ.get(_TEST_ISOLATION_MARKER_ENV)
)
#: pytest launcher names, matched against each argv token's *basename* so
#: ``/tmp/pytest-of-dev/...`` paths cannot false-positive.
_PYTEST_LAUNCHER_NAMES = frozenset({"pytest", "py.test", "pytest.exe", "py.test.exe"})
#: Memoised ancestry answer: the tree above us doesn't change; keep the hot path free.
_PYTEST_ANCESTOR: Optional[bool] = None
def _process_looks_like_pytest(proc: Any) -> bool:
"""True when *proc*'s command line is a pytest invocation (``pytest ...`` or
``python -m pytest``). Unreadable cmdline => not pytest: guessing the other
way would refuse production opens for unrelated reasons."""
try:
cmdline = proc.cmdline() or []
except Exception:
return False
for arg in cmdline:
try:
# Split on both separators on every host: os.path.basename is
# POSIX-only under Linux and would leave a Windows-style path
# intact, making the matcher's answer depend on the platform.
name = str(arg).strip('"').strip("'").replace("\\", "/").rsplit("/", 1)[-1].lower()
except Exception:
continue
if name in _PYTEST_LAUNCHER_NAMES:
return True
return False
def _has_pytest_ancestor() -> bool:
"""True when an ancestor process is a pytest run. A child spawned with a
rebuilt env loses PYTEST_* and the HERMES_HOME redirect together — aiming at
production AND disarming the guard in one step; ancestry survives that.
Fails open without psutil / on walk errors (never block real user runs)."""
global _PYTEST_ANCESTOR
if _PYTEST_ANCESTOR is not None:
return _PYTEST_ANCESTOR
found = False
if psutil is not None:
try:
found = any(_process_looks_like_pytest(p) for p in psutil.Process().parents())
except Exception:
found = False
_PYTEST_ANCESTOR = found
return found
def _in_test_context() -> bool:
"""Test run by environment or ancestry. Env first (two dict lookups); the
memoised ancestry walk runs at most once per real ``hermes`` invocation."""
return _running_under_pytest() or _has_pytest_ancestor()
def _production_state_roots() -> List[Path]:
roots: List[Path] = []
real_root = _real_platform_state_root()
@@ -710,19 +628,6 @@ def _production_state_roots() -> List[Path]:
return roots
def _is_production_state_db(resolved: Path, root: Path) -> bool:
"""*resolved* is ``<root>/state.db`` or ``<root>/profiles/<name>/state.db``.
Deeper scratch paths (repo worktrees under ~/.hermes/hermes-agent/...) are
deliberately NOT matched so hermetic tests cannot false-positive."""
if resolved.parent == root:
return True
try:
parts = resolved.relative_to(root).parts
except ValueError:
return False
return len(parts) == 3 and parts[0] == "profiles"
def _ensure_test_isolation(db_path: Path) -> None:
"""Raise RuntimeError before any connection/mkdir/pragma/byte probe when a
pytest-context process (env OR ancestry, see :func:`_in_test_context`)
@@ -749,27 +654,6 @@ def _ensure_test_isolation(db_path: Path) -> None:
)
# Last SessionDB() init error, per-process; surfaced by /resume-style slash
# commands so users know WHY. Only SessionDB.__init__ writes it (kanban_db
# failures are reported via their own callers, by design).
_last_init_error: Optional[str] = None
_last_init_error_lock = threading.Lock()
def _set_last_init_error(msg: Optional[str]) -> None:
"""Record (or clear with None) the most recent state.db init failure.
__init__ only SETs on failure and never clears on success: a concurrent
successful open would erase the cause another thread's /resume is about to format."""
global _last_init_error
with _last_init_error_lock:
_last_init_error = msg
def get_last_init_error() -> Optional[str]:
"""Most recent state.db init failure (None if none/never attempted)."""
return _last_init_error
# Openings of the background-review harness prompts (agent/background_review.py),
# matched case-sensitively against leading user/system content.
_REVIEW_HARNESS_PREFIXES = (
@@ -848,156 +732,12 @@ def format_session_db_unavailable(prefix: str = "Session database not available"
return f"{prefix}: {cause}{hint}."
# ---------------------------------------------------------------------------
# Malformed-schema recovery: ``sqlite_master`` itself is inconsistent (typically
# a DUPLICATE ``CREATE VIRTUAL TABLE messages_fts`` row). SQLite parses the
# whole schema while preparing the FIRST statement, so EVERY statement raises —
# including ``PRAGMA journal_mode`` (it trips in apply_wal_with_fallback during
# __init__, before _init_schema) and plain ``DROP TABLE``; only
# ``PRAGMA writable_schema=ON`` + sqlite_master surgery still work. Canonical
# sessions/messages are intact; recovery rebuilds only the FTS layer.
_MALFORMED_SCHEMA_MARKERS = ("malformed database schema",)
_MALFORMED_DB_MARKERS = (*_MALFORMED_SCHEMA_MARKERS, "database disk image is malformed")
# Auto-repair at most once per DB path per process (no repair loops; serialises
# concurrent web_server / gateway opens on the same malformed file).
_repair_attempted_paths: set[str] = set()
_repair_attempt_lock = threading.Lock()
def is_malformed_db_error(exc: BaseException) -> bool:
"""Malformed-schema OR generic corrupt-image error. Diagnostics / offline
recovery only — runtime repair must use :func:`is_malformed_schema_error`."""
return isinstance(exc, sqlite3.DatabaseError) and any(
marker in str(exc).lower() for marker in _MALFORMED_DB_MARKERS
)
# SQLITE_IOERR as a substring (wrapped strings still classify); shared by the
# read-only open retry and the write-path BEGIN retry.
_DISK_IO_ERROR_MARKER = "disk i/o error"
# "Store BUSY, not gone" — HTTP callers map these to 503 instead of 500.
# Corruption deliberately absent: a malformed store must surface, not be
# retried into a timeout.
_TRANSIENT_SQLITE_MARKERS = (
_DISK_IO_ERROR_MARKER, "database is locked", "database table is locked", "busy",
)
def _is_no_more_rows(exc: sqlite3.Error) -> bool:
"""Transient engine error on contended WAL appends; the identical write succeeds
standalone, so it retries like locked/busy. Message-scoped because some builds
raise it as InterfaceError (outside DatabaseError)."""
return "no more rows available" in str(exc).lower()
def is_transient_sqlite_error(exc: BaseException) -> bool:
""""Busy right now", not "damaged". One predicate so the read-only open
retry and the HTTP 503-vs-500 split cannot drift apart."""
return isinstance(exc, sqlite3.OperationalError) and any(
marker in str(exc).lower() for marker in _TRANSIENT_SQLITE_MARKERS
)
def _is_transient_read_only_ioerr(exc: sqlite3.OperationalError, *, attempt: int) -> bool:
"""Retry a read-only open? See _READ_ONLY_IOERR_RETRY_ATTEMPTS: a
persistent IOERR still exhausts the budget and propagates."""
return attempt < _READ_ONLY_IOERR_RETRY_ATTEMPTS and _DISK_IO_ERROR_MARKER in str(exc).lower()
def is_malformed_schema_error(exc: BaseException) -> bool:
"""Only SQLite's explicit malformed-schema text. A generic "disk image is
malformed" (SQLITE_CORRUPT) may be any B-tree/freelist page and does not
prove canonical rows intact, so runtime repair must fail closed on it."""
return isinstance(exc, sqlite3.DatabaseError) and any(
marker in str(exc).lower() for marker in _MALFORMED_SCHEMA_MARKERS
)
# "Filesystem cannot accept another write" substrings (OSError, sqlite3, and
# wrapped RPC strings all match the same helper).
_DISK_FULL_MARKERS = (
"no space left on device",
"not enough space",
"database or disk is full", # SQLITE_FULL
"disk full",
"full disk",
"enospc",
)
def is_disk_full_error(exc: BaseException | str | None) -> bool:
"""Disk-full / ENOSPC: OSError(ENOSPC), SQLITE_FULL, or matching strings."""
if exc is None:
return False
if isinstance(exc, OSError) and getattr(exc, "errno", None) == errno.ENOSPC:
return True
lowered = (exc if isinstance(exc, str) else str(exc)).lower()
return any(marker in lowered for marker in _DISK_FULL_MARKERS)
# Every classify_persistence_error bucket; consumers enumerate this tuple so a
# new bucket can never silently desynchronize them.
PERSISTENCE_ERROR_CAUSES = (
"locked", "compression", "compression_closed", "turn_lease", "corrupt", "replaced", "disk",
"unknown",
)
# "Database FILE structurally damaged" substrings. NOTE: "database disk image is
# malformed" contains "disk", so this check MUST run before the disk bucket in
# classify_persistence_error or B-tree corruption reads as "free some disk space".
_DB_CORRUPTION_MARKERS = (
"malformed", # "database disk image is malformed" (SQLITE_CORRUPT)
"file is not a database", # SQLITE_NOTADB (also connection-level poisoning)
"not a database",
"database corruption",
)
def classify_persistence_error(exc_or_str) -> str:
"""Coarse cause bucket (PERSISTENCE_ERROR_CAUSES) so the user's guidance
matches: "locked" = busy, retry; "disk" = full/read-only/permissions;
"compression" = a live lease refused the write; "compression_closed" = adopt
the rotated session id; "turn_lease" = fencing, not storage; "corrupt" =
file damage (repair path, not disk space); "replaced" = stop writing."""
if exc_or_str is None:
return "unknown"
# Lease refusals contain neither "locked" nor "busy": match by type, then by
# phrase for strings that survived RPC wrapping.
if isinstance(exc_or_str, SessionTurnLeaseLostError):
return "turn_lease"
if isinstance(exc_or_str, CompressionSessionClosedError):
return "compression_closed"
if isinstance(exc_or_str, CompressionSessionBusyError):
return "compression"
if isinstance(exc_or_str, StateDbReplacedError): # incl. DeletedWalGenerationError
return "replaced"
if isinstance(exc_or_str, StateDbCorruptError):
return "corrupt"
text = str(exc_or_str).lower()
if "turn lease" in text:
return "turn_lease"
if "closed by compression" in text:
return "compression_closed"
if "being compressed" in text or "compression lease" in text:
return "compression"
if "was replaced underneath" in text:
return "replaced"
if "deleted state.db-wal" in text or "deleted state.db-shm" in text:
return "replaced"
# Corruption BEFORE the lock/disk buckets: "disk image is malformed"
# contains "disk" and some wrapped strings mention "locked" recovery.
if any(marker in text for marker in _DB_CORRUPTION_MARKERS):
return "corrupt"
if "locked" in text or "busy" in text:
return "locked"
if is_disk_full_error(exc_or_str) or "disk" in text or "readonly" in text or "read-only" in text:
return "disk"
return "unknown"
# Cross-process schema-surgery lock: ``_repair_attempt_lock`` covers one
# interpreter only, while gateway, Desktop backend, CLI and TUI worker share the
# file and each used to run surgery + VACUUM on top of the winner's. Timeout
@@ -1123,87 +863,6 @@ def load_fts5_cjk_extension(conn: sqlite3.Connection) -> bool:
return False
class CompressionSessionClosedError(RuntimeError):
"""A durable write targeted a parent already closed by compression."""
def __init__(self, session_id: str):
self.session_id = session_id
super().__init__(
f"Session {session_id!r} is closed by compression; "
"adopt its live continuation before appending messages"
)
class CompressionSessionBusyError(RuntimeError):
"""A non-owner tried to write while compression owns the session."""
class SessionCompressionInProgressError(CompressionSessionBusyError):
"""A concurrent writer collided with a *live* compression lock — transient
(the compressor publishes in seconds; ``_execute_write`` waits), unlike the
parent class's other case (a compressor whose own lease is gone: permanent,
fail fast). Subclassing keeps every existing handler working."""
class SessionTurnLeaseLostError(RuntimeError):
"""A transcript write presented a turn-lease holder that no longer owns it.
Fail-fast fencing (no ``_execute_write`` retry): a later writer may already
be persisting a newer turn, and landing this one would interleave a stale reply."""
class StateDbReplacedError(RuntimeError):
"""The state.db path no longer names the file this SessionDB opened
(out-of-band cp/mv/restore). In-place FTS repair and fail-open trigger
dropping cannot fix a generation mismatch; they amplify it."""
class DeletedWalGenerationError(StateDbReplacedError):
"""A live process holds a deleted state.db-wal / -shm generation. Opening or
writing through this handle would mint a second WAL inode (split-brain ->
intermittent SQLITE_CORRUPT / IOERR). Stop the writers; never unlink the WAL
yourself. Subclasses StateDbReplacedError so every consumer that diverts
transcripts on a replaced store handles this identically."""
# SQLite header application_id (offset 68). Distinct from inode: ``cp`` onto the
# same path keeps st_ino and truncates+rewrites.
_STATE_DB_APPLICATION_ID_OFFSET = 68
_STATE_DB_GENERATION_KEY = "db_file_generation"
_STATE_DB_REPLACED_MSG = (
"FATAL: state.db was replaced underneath the gateway; refusing further "
"writes to this file. Divert transcripts to sessions/<id>.jsonl (and the "
"gateway pending_messages spool) and restore or reopen after operator intervention."
)
_DELETED_WAL_GENERATION_MSG = (
"FATAL: a live process holds a deleted state.db-wal or state.db-shm "
"inode while the path names a different (or missing) generation. "
"Refusing to open or write so a second WAL cannot be minted. "
"Stop the gateway, dashboard, and cron writers that hold the deleted "
"sidecar, then reopen. Do not delete the WAL yourself. "
"database.journal_mode: delete is operator containment, not a new default."
)
class StateDbCorruptError(sqlite3.DatabaseError):
"""A live SessionDB observed structural (non-FTS, non-replaced) corruption and
is quarantined: sticky for the handle's life — writes fail fast, no reopen,
no close-time checkpoint (a handle that kept writing after the first error
checkpointed 15 pages under wrong page numbers and turned a readable file
into "file is not a database"; SQLITE_DBCONFIG_NO_CKPT_ON_CLOSE on 3.12+
also stops SQLite's own). Subclasses sqlite3.DatabaseError so every degrade
path keeps working. Recovery boundary: restart on a repaired/restored file."""
_STATE_DB_CORRUPT_MSG = (
"FATAL: state.db reported structural corruption (database disk image is "
"malformed outside the FTS shadow tables) on a live handle; refusing further "
"writes, automatic reopen, and the close-time WAL checkpoint on this file. "
"Stop the gateway, then run `hermes sessions recover --source <state.db> "
"--inspect-only` or restore a snapshot. Unwritten transcripts are diverted to "
"sessions/<id>.jsonl (and the gateway pending_messages spool)."
)
def divert_session_transcript_jsonl(session_id: str, messages) -> "Optional[Path]":
"""Append pending messages to HERMES_HOME/sessions/<id>.jsonl (state.db was
replaced under a live process). Returns the path, or None if nothing to write."""

225
hermes_state_errors.py Normal file
View File

@@ -0,0 +1,225 @@
"""Exception types and error-classification predicates for the state store.
Shared by hermes_state and its mixins; string predicates match wrapped RPC
strings as well as live sqlite3 exceptions."""
import errno
import sqlite3
# ---------------------------------------------------------------------------
# Malformed-schema recovery: ``sqlite_master`` itself is inconsistent (typically
# a DUPLICATE ``CREATE VIRTUAL TABLE messages_fts`` row). SQLite parses the
# whole schema while preparing the FIRST statement, so EVERY statement raises —
# including ``PRAGMA journal_mode`` (it trips in apply_wal_with_fallback during
# __init__, before _init_schema) and plain ``DROP TABLE``; only
# ``PRAGMA writable_schema=ON`` + sqlite_master surgery still work. Canonical
# sessions/messages are intact; recovery rebuilds only the FTS layer.
_MALFORMED_SCHEMA_MARKERS = ("malformed database schema",)
_MALFORMED_DB_MARKERS = (*_MALFORMED_SCHEMA_MARKERS, "database disk image is malformed")
def is_malformed_db_error(exc: BaseException) -> bool:
"""Malformed-schema OR generic corrupt-image error. Diagnostics / offline
recovery only — runtime repair must use :func:`is_malformed_schema_error`."""
return isinstance(exc, sqlite3.DatabaseError) and any(
marker in str(exc).lower() for marker in _MALFORMED_DB_MARKERS
)
# SQLITE_IOERR as a substring (wrapped strings still classify); shared by the
# read-only open retry and the write-path BEGIN retry.
_DISK_IO_ERROR_MARKER = "disk i/o error"
# "Store BUSY, not gone" — HTTP callers map these to 503 instead of 500.
# Corruption deliberately absent: a malformed store must surface, not be
# retried into a timeout.
_TRANSIENT_SQLITE_MARKERS = (
_DISK_IO_ERROR_MARKER, "database is locked", "database table is locked", "busy",
)
def _is_no_more_rows(exc: sqlite3.Error) -> bool:
"""Transient engine error on contended WAL appends; the identical write succeeds
standalone, so it retries like locked/busy. Message-scoped because some builds
raise it as InterfaceError (outside DatabaseError)."""
return "no more rows available" in str(exc).lower()
def is_transient_sqlite_error(exc: BaseException) -> bool:
""""Busy right now", not "damaged". One predicate so the read-only open
retry and the HTTP 503-vs-500 split cannot drift apart."""
return isinstance(exc, sqlite3.OperationalError) and any(
marker in str(exc).lower() for marker in _TRANSIENT_SQLITE_MARKERS
)
def is_malformed_schema_error(exc: BaseException) -> bool:
"""Only SQLite's explicit malformed-schema text. A generic "disk image is
malformed" (SQLITE_CORRUPT) may be any B-tree/freelist page and does not
prove canonical rows intact, so runtime repair must fail closed on it."""
return isinstance(exc, sqlite3.DatabaseError) and any(
marker in str(exc).lower() for marker in _MALFORMED_SCHEMA_MARKERS
)
# "Filesystem cannot accept another write" substrings (OSError, sqlite3, and
# wrapped RPC strings all match the same helper).
_DISK_FULL_MARKERS = (
"no space left on device",
"not enough space",
"database or disk is full", # SQLITE_FULL
"disk full",
"full disk",
"enospc",
)
def is_disk_full_error(exc: BaseException | str | None) -> bool:
"""Disk-full / ENOSPC: OSError(ENOSPC), SQLITE_FULL, or matching strings."""
if exc is None:
return False
if isinstance(exc, OSError) and getattr(exc, "errno", None) == errno.ENOSPC:
return True
lowered = (exc if isinstance(exc, str) else str(exc)).lower()
return any(marker in lowered for marker in _DISK_FULL_MARKERS)
# Every classify_persistence_error bucket; consumers enumerate this tuple so a
# new bucket can never silently desynchronize them.
PERSISTENCE_ERROR_CAUSES = (
"locked", "compression", "compression_closed", "turn_lease", "corrupt", "replaced", "disk",
"unknown",
)
# "Database FILE structurally damaged" substrings. NOTE: "database disk image is
# malformed" contains "disk", so this check MUST run before the disk bucket in
# classify_persistence_error or B-tree corruption reads as "free some disk space".
_DB_CORRUPTION_MARKERS = (
"malformed", # "database disk image is malformed" (SQLITE_CORRUPT)
"file is not a database", # SQLITE_NOTADB (also connection-level poisoning)
"not a database",
"database corruption",
)
def classify_persistence_error(exc_or_str) -> str:
"""Coarse cause bucket (PERSISTENCE_ERROR_CAUSES) so the user's guidance
matches: "locked" = busy, retry; "disk" = full/read-only/permissions;
"compression" = a live lease refused the write; "compression_closed" = adopt
the rotated session id; "turn_lease" = fencing, not storage; "corrupt" =
file damage (repair path, not disk space); "replaced" = stop writing."""
if exc_or_str is None:
return "unknown"
# Lease refusals contain neither "locked" nor "busy": match by type, then by
# phrase for strings that survived RPC wrapping.
if isinstance(exc_or_str, SessionTurnLeaseLostError):
return "turn_lease"
if isinstance(exc_or_str, CompressionSessionClosedError):
return "compression_closed"
if isinstance(exc_or_str, CompressionSessionBusyError):
return "compression"
if isinstance(exc_or_str, StateDbReplacedError): # incl. DeletedWalGenerationError
return "replaced"
if isinstance(exc_or_str, StateDbCorruptError):
return "corrupt"
text = str(exc_or_str).lower()
if "turn lease" in text:
return "turn_lease"
if "closed by compression" in text:
return "compression_closed"
if "being compressed" in text or "compression lease" in text:
return "compression"
if "was replaced underneath" in text:
return "replaced"
if "deleted state.db-wal" in text or "deleted state.db-shm" in text:
return "replaced"
# Corruption BEFORE the lock/disk buckets: "disk image is malformed"
# contains "disk" and some wrapped strings mention "locked" recovery.
if any(marker in text for marker in _DB_CORRUPTION_MARKERS):
return "corrupt"
if "locked" in text or "busy" in text:
return "locked"
if is_disk_full_error(exc_or_str) or "disk" in text or "readonly" in text or "read-only" in text:
return "disk"
return "unknown"
class CompressionSessionClosedError(RuntimeError):
"""A durable write targeted a parent already closed by compression."""
def __init__(self, session_id: str):
self.session_id = session_id
super().__init__(
f"Session {session_id!r} is closed by compression; "
"adopt its live continuation before appending messages"
)
class CompressionSessionBusyError(RuntimeError):
"""A non-owner tried to write while compression owns the session."""
class SessionCompressionInProgressError(CompressionSessionBusyError):
"""A concurrent writer collided with a *live* compression lock — transient
(the compressor publishes in seconds; ``_execute_write`` waits), unlike the
parent class's other case (a compressor whose own lease is gone: permanent,
fail fast). Subclassing keeps every existing handler working."""
class SessionTurnLeaseLostError(RuntimeError):
"""A transcript write presented a turn-lease holder that no longer owns it.
Fail-fast fencing (no ``_execute_write`` retry): a later writer may already
be persisting a newer turn, and landing this one would interleave a stale reply."""
class StateDbReplacedError(RuntimeError):
"""The state.db path no longer names the file this SessionDB opened
(out-of-band cp/mv/restore). In-place FTS repair and fail-open trigger
dropping cannot fix a generation mismatch; they amplify it."""
class DeletedWalGenerationError(StateDbReplacedError):
"""A live process holds a deleted state.db-wal / -shm generation. Opening or
writing through this handle would mint a second WAL inode (split-brain ->
intermittent SQLITE_CORRUPT / IOERR). Stop the writers; never unlink the WAL
yourself. Subclasses StateDbReplacedError so every consumer that diverts
transcripts on a replaced store handles this identically."""
# SQLite header application_id (offset 68). Distinct from inode: ``cp`` onto the
# same path keeps st_ino and truncates+rewrites.
_STATE_DB_APPLICATION_ID_OFFSET = 68
_STATE_DB_GENERATION_KEY = "db_file_generation"
_STATE_DB_REPLACED_MSG = (
"FATAL: state.db was replaced underneath the gateway; refusing further "
"writes to this file. Divert transcripts to sessions/<id>.jsonl (and the "
"gateway pending_messages spool) and restore or reopen after operator intervention."
)
_DELETED_WAL_GENERATION_MSG = (
"FATAL: a live process holds a deleted state.db-wal or state.db-shm "
"inode while the path names a different (or missing) generation. "
"Refusing to open or write so a second WAL cannot be minted. "
"Stop the gateway, dashboard, and cron writers that hold the deleted "
"sidecar, then reopen. Do not delete the WAL yourself. "
"database.journal_mode: delete is operator containment, not a new default."
)
class StateDbCorruptError(sqlite3.DatabaseError):
"""A live SessionDB observed structural (non-FTS, non-replaced) corruption and
is quarantined: sticky for the handle's life — writes fail fast, no reopen,
no close-time checkpoint (a handle that kept writing after the first error
checkpointed 15 pages under wrong page numbers and turned a readable file
into "file is not a database"; SQLITE_DBCONFIG_NO_CKPT_ON_CLOSE on 3.12+
also stops SQLite's own). Subclasses sqlite3.DatabaseError so every degrade
path keeps working. Recovery boundary: restart on a repaired/restored file."""
_STATE_DB_CORRUPT_MSG = (
"FATAL: state.db reported structural corruption (database disk image is "
"malformed outside the FTS shadow tables) on a live handle; refusing further "
"writes, automatic reopen, and the close-time WAL checkpoint on this file. "
"Stop the gateway, then run `hermes sessions recover --source <state.db> "
"--inspect-only` or restore a snapshot. Unwritten transcripts are diverted to "
"sessions/<id>.jsonl (and the gateway pending_messages spool)."
)

148
hermes_state_guard.py Normal file
View File

@@ -0,0 +1,148 @@
"""Live-DB test-isolation guard and the per-process "last init error" record.
Every SessionDB construction resolves its path through _ensure_test_isolation
so a pytest-context process (env OR ancestry) can never open a production
state.db; env-based so subprocess children are protected too."""
import os
import sys
import threading
from pathlib import Path
from typing import Any, Optional
try: # Hard dependency, but tolerate scaffold-phase imports before pip install.
import psutil
except ImportError: # pragma: no cover - stripped/scaffold installs only
psutil = None # type: ignore[assignment]
# Field evidence: pytest fixture rows landed in the production state.db and a
# pytest-spawned child flipped the journal mode under the live WAL writer,
# destroying committed transcripts; any HERMES_HOME escape (fixture ordering, a
# child spawned without it, a shell exporting the real home) fell through silently.
#: Env twin of ``_STATE_DB_GUARD_BYPASS`` for child processes (a module global
#: cannot cross a process boundary, and ancestry arms the guard there).
_STATE_DB_GUARD_BYPASS_ENV = "HERMES_STATE_DB_GUARD_BYPASS"
def _real_platform_state_root() -> Optional[Path]:
"""The REAL platform-default Hermes root. Avoids ``Path.home()`` /
``hermes_constants``: tests monkeypatch Path.home to a tempdir while this
module is imported lazily, which would misidentify the hermetic home as
production or miss the real one. ``expanduser`` reads HOME/passwd, which the
conftest never rewrites."""
try:
if sys.platform == "win32":
base = os.environ.get("LOCALAPPDATA", "").strip()
root = (
Path(base) / "hermes"
if base
else Path(os.path.expanduser("~")) / "AppData" / "Local" / "hermes"
)
else:
root = Path(os.path.expanduser("~")) / ".hermes"
return root.resolve()
except Exception:
return None
#: Exported by the hermetic conftest alongside the HERMES_HOME redirect (value:
#: the isolation root). Unlike PYTEST_* (scrubbed by tests that rebuild a child
#: env) it is OURS and inherits by default, so a child carrying it that resolves
#: a production DB is by definition an isolation escape.
_TEST_ISOLATION_MARKER_ENV = "HERMES_TEST_ISOLATION"
def _running_under_pytest() -> bool:
"""True when this process (or a parent test process) is a pytest run."""
return bool(
os.environ.get("PYTEST_CURRENT_TEST")
or os.environ.get("PYTEST_VERSION")
or os.environ.get(_TEST_ISOLATION_MARKER_ENV)
)
#: pytest launcher names, matched against each argv token's *basename* so
#: ``/tmp/pytest-of-dev/...`` paths cannot false-positive.
_PYTEST_LAUNCHER_NAMES = frozenset({"pytest", "py.test", "pytest.exe", "py.test.exe"})
#: Memoised ancestry answer: the tree above us doesn't change; keep the hot path free.
_PYTEST_ANCESTOR: Optional[bool] = None
def _process_looks_like_pytest(proc: Any) -> bool:
"""True when *proc*'s command line is a pytest invocation (``pytest ...`` or
``python -m pytest``). Unreadable cmdline => not pytest: guessing the other
way would refuse production opens for unrelated reasons."""
try:
cmdline = proc.cmdline() or []
except Exception:
return False
for arg in cmdline:
try:
# Split on both separators on every host: os.path.basename is
# POSIX-only under Linux and would leave a Windows-style path
# intact, making the matcher's answer depend on the platform.
name = str(arg).strip('"').strip("'").replace("\\", "/").rsplit("/", 1)[-1].lower()
except Exception:
continue
if name in _PYTEST_LAUNCHER_NAMES:
return True
return False
def _has_pytest_ancestor() -> bool:
"""True when an ancestor process is a pytest run. A child spawned with a
rebuilt env loses PYTEST_* and the HERMES_HOME redirect together — aiming at
production AND disarming the guard in one step; ancestry survives that.
Fails open without psutil / on walk errors (never block real user runs)."""
global _PYTEST_ANCESTOR
if _PYTEST_ANCESTOR is not None:
return _PYTEST_ANCESTOR
found = False
if psutil is not None:
try:
found = any(_process_looks_like_pytest(p) for p in psutil.Process().parents())
except Exception:
found = False
_PYTEST_ANCESTOR = found
return found
def _in_test_context() -> bool:
"""Test run by environment or ancestry. Env first (two dict lookups); the
memoised ancestry walk runs at most once per real ``hermes`` invocation."""
return _running_under_pytest() or _has_pytest_ancestor()
def _is_production_state_db(resolved: Path, root: Path) -> bool:
"""*resolved* is ``<root>/state.db`` or ``<root>/profiles/<name>/state.db``.
Deeper scratch paths (repo worktrees under ~/.hermes/hermes-agent/...) are
deliberately NOT matched so hermetic tests cannot false-positive."""
if resolved.parent == root:
return True
try:
parts = resolved.relative_to(root).parts
except ValueError:
return False
return len(parts) == 3 and parts[0] == "profiles"
# Last SessionDB() init error, per-process; surfaced by /resume-style slash
# commands so users know WHY. Only SessionDB.__init__ writes it (kanban_db
# failures are reported via their own callers, by design).
_last_init_error: Optional[str] = None
_last_init_error_lock = threading.Lock()
def _set_last_init_error(msg: Optional[str]) -> None:
"""Record (or clear with None) the most recent state.db init failure.
__init__ only SETs on failure and never clears on success: a concurrent
successful open would erase the cause another thread's /resume is about to format."""
global _last_init_error
with _last_init_error_lock:
_last_init_error = msg
def get_last_init_error() -> Optional[str]:
"""Most recent state.db init failure (None if none/never attempted)."""
return _last_init_error