diff --git a/hermes_state.py b/hermes_state.py index 8fe677a42a..728c609299 100644 --- a/hermes_state.py +++ b/hermes_state.py @@ -45,6 +45,21 @@ from hermes_state_common import ( # noqa: F401 (re-exported; tests import from FTS_TRIGRAM_SQL, LEGACY_FTS_SQL, LEGACY_FTS_TRIGRAM_SQL, SCHEMA_SQL, SCHEMA_VERSION, stat_db_file_identity as _stat_db_file_identity, ) +from hermes_state_errors import ( # noqa: F401 (re-exported; the historical import path) + _DB_CORRUPTION_MARKERS, _DELETED_WAL_GENERATION_MSG, _DISK_FULL_MARKERS, _DISK_IO_ERROR_MARKER, + _MALFORMED_DB_MARKERS, _MALFORMED_SCHEMA_MARKERS, _STATE_DB_APPLICATION_ID_OFFSET, + _STATE_DB_CORRUPT_MSG, _STATE_DB_GENERATION_KEY, _STATE_DB_REPLACED_MSG, + _TRANSIENT_SQLITE_MARKERS, PERSISTENCE_ERROR_CAUSES, CompressionSessionBusyError, + CompressionSessionClosedError, DeletedWalGenerationError, SessionCompressionInProgressError, + SessionTurnLeaseLostError, StateDbCorruptError, StateDbReplacedError, _is_no_more_rows, + classify_persistence_error, is_disk_full_error, is_malformed_db_error, + is_malformed_schema_error, is_transient_sqlite_error, +) +from hermes_state_guard import ( # noqa: F401 (re-exported; tests patch hermes_state.) + _PYTEST_LAUNCHER_NAMES, _STATE_DB_GUARD_BYPASS_ENV, _TEST_ISOLATION_MARKER_ENV, + _has_pytest_ancestor, _in_test_context, _is_production_state_db, _process_looks_like_pytest, + _real_platform_state_root, _running_under_pytest, _set_last_init_error, get_last_init_error, +) from hermes_state_portability import SessionPortabilityMixin from hermes_state_telegram import SessionTelegramTopicsMixin from hermes_state_schema import SessionSchemaMixin @@ -361,6 +376,12 @@ _READ_OPEN_RETRY_SECONDS = 60.0 _READ_ONLY_IOERR_RETRY_ATTEMPTS = 3 _READ_ONLY_IOERR_RETRY_BACKOFF_S = 0.05 +def _is_transient_read_only_ioerr(exc: sqlite3.OperationalError, *, attempt: int) -> bool: + """Retry a read-only open? See _READ_ONLY_IOERR_RETRY_ATTEMPTS: a + persistent IOERR still exhausts the budget and propagates.""" + return attempt < _READ_ONLY_IOERR_RETRY_ATTEMPTS and _DISK_IO_ERROR_MARKER in str(exc).lower() + + # Ceiling on read-only connections ALIVE at once against one database FILE # (idle pooled + checked out, summed over every SessionDB on that file). One # constant for both the pool maxsize and the permit count: a LifoQueue only caps @@ -583,120 +604,17 @@ def _default_db_path() -> Path: return get_hermes_home() / "state.db" -# --------------------------------------------------------------------------- -# Live-DB test-isolation guard -# --------------------------------------------------------------------------- -# Field evidence: pytest fixture rows landed in the production state.db and a -# pytest-spawned child flipped the journal mode under the live WAL writer, -# destroying committed transcripts. The hermetic conftest redirects HERMES_HOME -# per test, but any escape (fixture ordering, a child spawned without -# HERMES_HOME, a shell exporting the real home) fell through silently. EVERY -# SessionDB construction resolves its path here, so under pytest a production -# state.db fails hard. Env-based, so subprocess children are protected too. - +# Live-DB guard knobs live HERE (not in hermes_state_guard): the hermetic conftest +# monkeypatches ``hermes_state._STATE_DB_GUARD_BYPASS`` / ``_EXTRA_DENY_ROOTS``. #: Escape hatch for tests that genuinely need the real DB (conftest sets it for #: ``@pytest.mark.live_system_guard_bypass``); scripts may set it explicitly. _STATE_DB_GUARD_BYPASS = False -#: Env twin of ``_STATE_DB_GUARD_BYPASS`` for child processes (a module global -#: cannot cross a process boundary, and ancestry arms the guard there). -_STATE_DB_GUARD_BYPASS_ENV = "HERMES_STATE_DB_GUARD_BYPASS" - #: Extra production roots to refuse; conftest injects the pre-sandbox root so #: custom-HERMES_HOME deployments are covered too. _STATE_DB_GUARD_EXTRA_DENY_ROOTS: Tuple[Path, ...] = () -def _real_platform_state_root() -> Optional[Path]: - """The REAL platform-default Hermes root. Avoids ``Path.home()`` / - ``hermes_constants``: tests monkeypatch Path.home to a tempdir while this - module is imported lazily, which would misidentify the hermetic home as - production or miss the real one. ``expanduser`` reads HOME/passwd, which the - conftest never rewrites.""" - try: - if sys.platform == "win32": - base = os.environ.get("LOCALAPPDATA", "").strip() - root = ( - Path(base) / "hermes" - if base - else Path(os.path.expanduser("~")) / "AppData" / "Local" / "hermes" - ) - else: - root = Path(os.path.expanduser("~")) / ".hermes" - return root.resolve() - except Exception: - return None - - -#: Exported by the hermetic conftest alongside the HERMES_HOME redirect (value: -#: the isolation root). Unlike PYTEST_* (scrubbed by tests that rebuild a child -#: env) it is OURS and inherits by default, so a child carrying it that resolves -#: a production DB is by definition an isolation escape. -_TEST_ISOLATION_MARKER_ENV = "HERMES_TEST_ISOLATION" - - -def _running_under_pytest() -> bool: - """True when this process (or a parent test process) is a pytest run.""" - return bool( - os.environ.get("PYTEST_CURRENT_TEST") - or os.environ.get("PYTEST_VERSION") - or os.environ.get(_TEST_ISOLATION_MARKER_ENV) - ) - - -#: pytest launcher names, matched against each argv token's *basename* so -#: ``/tmp/pytest-of-dev/...`` paths cannot false-positive. -_PYTEST_LAUNCHER_NAMES = frozenset({"pytest", "py.test", "pytest.exe", "py.test.exe"}) - -#: Memoised ancestry answer: the tree above us doesn't change; keep the hot path free. -_PYTEST_ANCESTOR: Optional[bool] = None - - -def _process_looks_like_pytest(proc: Any) -> bool: - """True when *proc*'s command line is a pytest invocation (``pytest ...`` or - ``python -m pytest``). Unreadable cmdline => not pytest: guessing the other - way would refuse production opens for unrelated reasons.""" - try: - cmdline = proc.cmdline() or [] - except Exception: - return False - for arg in cmdline: - try: - # Split on both separators on every host: os.path.basename is - # POSIX-only under Linux and would leave a Windows-style path - # intact, making the matcher's answer depend on the platform. - name = str(arg).strip('"').strip("'").replace("\\", "/").rsplit("/", 1)[-1].lower() - except Exception: - continue - if name in _PYTEST_LAUNCHER_NAMES: - return True - return False - - -def _has_pytest_ancestor() -> bool: - """True when an ancestor process is a pytest run. A child spawned with a - rebuilt env loses PYTEST_* and the HERMES_HOME redirect together — aiming at - production AND disarming the guard in one step; ancestry survives that. - Fails open without psutil / on walk errors (never block real user runs).""" - global _PYTEST_ANCESTOR - if _PYTEST_ANCESTOR is not None: - return _PYTEST_ANCESTOR - found = False - if psutil is not None: - try: - found = any(_process_looks_like_pytest(p) for p in psutil.Process().parents()) - except Exception: - found = False - _PYTEST_ANCESTOR = found - return found - - -def _in_test_context() -> bool: - """Test run by environment or ancestry. Env first (two dict lookups); the - memoised ancestry walk runs at most once per real ``hermes`` invocation.""" - return _running_under_pytest() or _has_pytest_ancestor() - - def _production_state_roots() -> List[Path]: roots: List[Path] = [] real_root = _real_platform_state_root() @@ -710,19 +628,6 @@ def _production_state_roots() -> List[Path]: return roots -def _is_production_state_db(resolved: Path, root: Path) -> bool: - """*resolved* is ``/state.db`` or ``/profiles//state.db``. - Deeper scratch paths (repo worktrees under ~/.hermes/hermes-agent/...) are - deliberately NOT matched so hermetic tests cannot false-positive.""" - if resolved.parent == root: - return True - try: - parts = resolved.relative_to(root).parts - except ValueError: - return False - return len(parts) == 3 and parts[0] == "profiles" - - def _ensure_test_isolation(db_path: Path) -> None: """Raise RuntimeError before any connection/mkdir/pragma/byte probe when a pytest-context process (env OR ancestry, see :func:`_in_test_context`) @@ -749,27 +654,6 @@ def _ensure_test_isolation(db_path: Path) -> None: ) -# Last SessionDB() init error, per-process; surfaced by /resume-style slash -# commands so users know WHY. Only SessionDB.__init__ writes it (kanban_db -# failures are reported via their own callers, by design). -_last_init_error: Optional[str] = None -_last_init_error_lock = threading.Lock() - - -def _set_last_init_error(msg: Optional[str]) -> None: - """Record (or clear with None) the most recent state.db init failure. - __init__ only SETs on failure and never clears on success: a concurrent - successful open would erase the cause another thread's /resume is about to format.""" - global _last_init_error - with _last_init_error_lock: - _last_init_error = msg - - -def get_last_init_error() -> Optional[str]: - """Most recent state.db init failure (None if none/never attempted).""" - return _last_init_error - - # Openings of the background-review harness prompts (agent/background_review.py), # matched case-sensitively against leading user/system content. _REVIEW_HARNESS_PREFIXES = ( @@ -848,156 +732,12 @@ def format_session_db_unavailable(prefix: str = "Session database not available" return f"{prefix}: {cause}{hint}." -# --------------------------------------------------------------------------- -# Malformed-schema recovery: ``sqlite_master`` itself is inconsistent (typically -# a DUPLICATE ``CREATE VIRTUAL TABLE messages_fts`` row). SQLite parses the -# whole schema while preparing the FIRST statement, so EVERY statement raises — -# including ``PRAGMA journal_mode`` (it trips in apply_wal_with_fallback during -# __init__, before _init_schema) and plain ``DROP TABLE``; only -# ``PRAGMA writable_schema=ON`` + sqlite_master surgery still work. Canonical -# sessions/messages are intact; recovery rebuilds only the FTS layer. -_MALFORMED_SCHEMA_MARKERS = ("malformed database schema",) -_MALFORMED_DB_MARKERS = (*_MALFORMED_SCHEMA_MARKERS, "database disk image is malformed") - # Auto-repair at most once per DB path per process (no repair loops; serialises # concurrent web_server / gateway opens on the same malformed file). _repair_attempted_paths: set[str] = set() _repair_attempt_lock = threading.Lock() -def is_malformed_db_error(exc: BaseException) -> bool: - """Malformed-schema OR generic corrupt-image error. Diagnostics / offline - recovery only — runtime repair must use :func:`is_malformed_schema_error`.""" - return isinstance(exc, sqlite3.DatabaseError) and any( - marker in str(exc).lower() for marker in _MALFORMED_DB_MARKERS - ) - - -# SQLITE_IOERR as a substring (wrapped strings still classify); shared by the -# read-only open retry and the write-path BEGIN retry. -_DISK_IO_ERROR_MARKER = "disk i/o error" - -# "Store BUSY, not gone" — HTTP callers map these to 503 instead of 500. -# Corruption deliberately absent: a malformed store must surface, not be -# retried into a timeout. -_TRANSIENT_SQLITE_MARKERS = ( - _DISK_IO_ERROR_MARKER, "database is locked", "database table is locked", "busy", -) - - -def _is_no_more_rows(exc: sqlite3.Error) -> bool: - """Transient engine error on contended WAL appends; the identical write succeeds - standalone, so it retries like locked/busy. Message-scoped because some builds - raise it as InterfaceError (outside DatabaseError).""" - return "no more rows available" in str(exc).lower() - - -def is_transient_sqlite_error(exc: BaseException) -> bool: - """"Busy right now", not "damaged". One predicate so the read-only open - retry and the HTTP 503-vs-500 split cannot drift apart.""" - return isinstance(exc, sqlite3.OperationalError) and any( - marker in str(exc).lower() for marker in _TRANSIENT_SQLITE_MARKERS - ) - - -def _is_transient_read_only_ioerr(exc: sqlite3.OperationalError, *, attempt: int) -> bool: - """Retry a read-only open? See _READ_ONLY_IOERR_RETRY_ATTEMPTS: a - persistent IOERR still exhausts the budget and propagates.""" - return attempt < _READ_ONLY_IOERR_RETRY_ATTEMPTS and _DISK_IO_ERROR_MARKER in str(exc).lower() - - -def is_malformed_schema_error(exc: BaseException) -> bool: - """Only SQLite's explicit malformed-schema text. A generic "disk image is - malformed" (SQLITE_CORRUPT) may be any B-tree/freelist page and does not - prove canonical rows intact, so runtime repair must fail closed on it.""" - return isinstance(exc, sqlite3.DatabaseError) and any( - marker in str(exc).lower() for marker in _MALFORMED_SCHEMA_MARKERS - ) - - -# "Filesystem cannot accept another write" substrings (OSError, sqlite3, and -# wrapped RPC strings all match the same helper). -_DISK_FULL_MARKERS = ( - "no space left on device", - "not enough space", - "database or disk is full", # SQLITE_FULL - "disk full", - "full disk", - "enospc", -) - - -def is_disk_full_error(exc: BaseException | str | None) -> bool: - """Disk-full / ENOSPC: OSError(ENOSPC), SQLITE_FULL, or matching strings.""" - if exc is None: - return False - if isinstance(exc, OSError) and getattr(exc, "errno", None) == errno.ENOSPC: - return True - lowered = (exc if isinstance(exc, str) else str(exc)).lower() - return any(marker in lowered for marker in _DISK_FULL_MARKERS) - - -# Every classify_persistence_error bucket; consumers enumerate this tuple so a -# new bucket can never silently desynchronize them. -PERSISTENCE_ERROR_CAUSES = ( - "locked", "compression", "compression_closed", "turn_lease", "corrupt", "replaced", "disk", - "unknown", -) - - -# "Database FILE structurally damaged" substrings. NOTE: "database disk image is -# malformed" contains "disk", so this check MUST run before the disk bucket in -# classify_persistence_error or B-tree corruption reads as "free some disk space". -_DB_CORRUPTION_MARKERS = ( - "malformed", # "database disk image is malformed" (SQLITE_CORRUPT) - "file is not a database", # SQLITE_NOTADB (also connection-level poisoning) - "not a database", - "database corruption", -) - - -def classify_persistence_error(exc_or_str) -> str: - """Coarse cause bucket (PERSISTENCE_ERROR_CAUSES) so the user's guidance - matches: "locked" = busy, retry; "disk" = full/read-only/permissions; - "compression" = a live lease refused the write; "compression_closed" = adopt - the rotated session id; "turn_lease" = fencing, not storage; "corrupt" = - file damage (repair path, not disk space); "replaced" = stop writing.""" - if exc_or_str is None: - return "unknown" - # Lease refusals contain neither "locked" nor "busy": match by type, then by - # phrase for strings that survived RPC wrapping. - if isinstance(exc_or_str, SessionTurnLeaseLostError): - return "turn_lease" - if isinstance(exc_or_str, CompressionSessionClosedError): - return "compression_closed" - if isinstance(exc_or_str, CompressionSessionBusyError): - return "compression" - if isinstance(exc_or_str, StateDbReplacedError): # incl. DeletedWalGenerationError - return "replaced" - if isinstance(exc_or_str, StateDbCorruptError): - return "corrupt" - text = str(exc_or_str).lower() - if "turn lease" in text: - return "turn_lease" - if "closed by compression" in text: - return "compression_closed" - if "being compressed" in text or "compression lease" in text: - return "compression" - if "was replaced underneath" in text: - return "replaced" - if "deleted state.db-wal" in text or "deleted state.db-shm" in text: - return "replaced" - # Corruption BEFORE the lock/disk buckets: "disk image is malformed" - # contains "disk" and some wrapped strings mention "locked" recovery. - if any(marker in text for marker in _DB_CORRUPTION_MARKERS): - return "corrupt" - if "locked" in text or "busy" in text: - return "locked" - if is_disk_full_error(exc_or_str) or "disk" in text or "readonly" in text or "read-only" in text: - return "disk" - return "unknown" - - # Cross-process schema-surgery lock: ``_repair_attempt_lock`` covers one # interpreter only, while gateway, Desktop backend, CLI and TUI worker share the # file and each used to run surgery + VACUUM on top of the winner's. Timeout @@ -1123,87 +863,6 @@ def load_fts5_cjk_extension(conn: sqlite3.Connection) -> bool: return False -class CompressionSessionClosedError(RuntimeError): - """A durable write targeted a parent already closed by compression.""" - - def __init__(self, session_id: str): - self.session_id = session_id - super().__init__( - f"Session {session_id!r} is closed by compression; " - "adopt its live continuation before appending messages" - ) - - -class CompressionSessionBusyError(RuntimeError): - """A non-owner tried to write while compression owns the session.""" - - -class SessionCompressionInProgressError(CompressionSessionBusyError): - """A concurrent writer collided with a *live* compression lock — transient - (the compressor publishes in seconds; ``_execute_write`` waits), unlike the - parent class's other case (a compressor whose own lease is gone: permanent, - fail fast). Subclassing keeps every existing handler working.""" - - -class SessionTurnLeaseLostError(RuntimeError): - """A transcript write presented a turn-lease holder that no longer owns it. - Fail-fast fencing (no ``_execute_write`` retry): a later writer may already - be persisting a newer turn, and landing this one would interleave a stale reply.""" - - -class StateDbReplacedError(RuntimeError): - """The state.db path no longer names the file this SessionDB opened - (out-of-band cp/mv/restore). In-place FTS repair and fail-open trigger - dropping cannot fix a generation mismatch; they amplify it.""" - - -class DeletedWalGenerationError(StateDbReplacedError): - """A live process holds a deleted state.db-wal / -shm generation. Opening or - writing through this handle would mint a second WAL inode (split-brain -> - intermittent SQLITE_CORRUPT / IOERR). Stop the writers; never unlink the WAL - yourself. Subclasses StateDbReplacedError so every consumer that diverts - transcripts on a replaced store handles this identically.""" - - -# SQLite header application_id (offset 68). Distinct from inode: ``cp`` onto the -# same path keeps st_ino and truncates+rewrites. -_STATE_DB_APPLICATION_ID_OFFSET = 68 -_STATE_DB_GENERATION_KEY = "db_file_generation" -_STATE_DB_REPLACED_MSG = ( - "FATAL: state.db was replaced underneath the gateway; refusing further " - "writes to this file. Divert transcripts to sessions/.jsonl (and the " - "gateway pending_messages spool) and restore or reopen after operator intervention." -) -_DELETED_WAL_GENERATION_MSG = ( - "FATAL: a live process holds a deleted state.db-wal or state.db-shm " - "inode while the path names a different (or missing) generation. " - "Refusing to open or write so a second WAL cannot be minted. " - "Stop the gateway, dashboard, and cron writers that hold the deleted " - "sidecar, then reopen. Do not delete the WAL yourself. " - "database.journal_mode: delete is operator containment, not a new default." -) - - -class StateDbCorruptError(sqlite3.DatabaseError): - """A live SessionDB observed structural (non-FTS, non-replaced) corruption and - is quarantined: sticky for the handle's life — writes fail fast, no reopen, - no close-time checkpoint (a handle that kept writing after the first error - checkpointed 15 pages under wrong page numbers and turned a readable file - into "file is not a database"; SQLITE_DBCONFIG_NO_CKPT_ON_CLOSE on 3.12+ - also stops SQLite's own). Subclasses sqlite3.DatabaseError so every degrade - path keeps working. Recovery boundary: restart on a repaired/restored file.""" - - -_STATE_DB_CORRUPT_MSG = ( - "FATAL: state.db reported structural corruption (database disk image is " - "malformed outside the FTS shadow tables) on a live handle; refusing further " - "writes, automatic reopen, and the close-time WAL checkpoint on this file. " - "Stop the gateway, then run `hermes sessions recover --source " - "--inspect-only` or restore a snapshot. Unwritten transcripts are diverted to " - "sessions/.jsonl (and the gateway pending_messages spool)." -) - - def divert_session_transcript_jsonl(session_id: str, messages) -> "Optional[Path]": """Append pending messages to HERMES_HOME/sessions/.jsonl (state.db was replaced under a live process). Returns the path, or None if nothing to write.""" diff --git a/hermes_state_errors.py b/hermes_state_errors.py new file mode 100644 index 0000000000..806c81b7ea --- /dev/null +++ b/hermes_state_errors.py @@ -0,0 +1,225 @@ +"""Exception types and error-classification predicates for the state store. +Shared by hermes_state and its mixins; string predicates match wrapped RPC +strings as well as live sqlite3 exceptions.""" + +import errno +import sqlite3 + +# --------------------------------------------------------------------------- +# Malformed-schema recovery: ``sqlite_master`` itself is inconsistent (typically +# a DUPLICATE ``CREATE VIRTUAL TABLE messages_fts`` row). SQLite parses the +# whole schema while preparing the FIRST statement, so EVERY statement raises — +# including ``PRAGMA journal_mode`` (it trips in apply_wal_with_fallback during +# __init__, before _init_schema) and plain ``DROP TABLE``; only +# ``PRAGMA writable_schema=ON`` + sqlite_master surgery still work. Canonical +# sessions/messages are intact; recovery rebuilds only the FTS layer. +_MALFORMED_SCHEMA_MARKERS = ("malformed database schema",) +_MALFORMED_DB_MARKERS = (*_MALFORMED_SCHEMA_MARKERS, "database disk image is malformed") + + +def is_malformed_db_error(exc: BaseException) -> bool: + """Malformed-schema OR generic corrupt-image error. Diagnostics / offline + recovery only — runtime repair must use :func:`is_malformed_schema_error`.""" + return isinstance(exc, sqlite3.DatabaseError) and any( + marker in str(exc).lower() for marker in _MALFORMED_DB_MARKERS + ) + + +# SQLITE_IOERR as a substring (wrapped strings still classify); shared by the +# read-only open retry and the write-path BEGIN retry. +_DISK_IO_ERROR_MARKER = "disk i/o error" + +# "Store BUSY, not gone" — HTTP callers map these to 503 instead of 500. +# Corruption deliberately absent: a malformed store must surface, not be +# retried into a timeout. +_TRANSIENT_SQLITE_MARKERS = ( + _DISK_IO_ERROR_MARKER, "database is locked", "database table is locked", "busy", +) + + +def _is_no_more_rows(exc: sqlite3.Error) -> bool: + """Transient engine error on contended WAL appends; the identical write succeeds + standalone, so it retries like locked/busy. Message-scoped because some builds + raise it as InterfaceError (outside DatabaseError).""" + return "no more rows available" in str(exc).lower() + + +def is_transient_sqlite_error(exc: BaseException) -> bool: + """"Busy right now", not "damaged". One predicate so the read-only open + retry and the HTTP 503-vs-500 split cannot drift apart.""" + return isinstance(exc, sqlite3.OperationalError) and any( + marker in str(exc).lower() for marker in _TRANSIENT_SQLITE_MARKERS + ) + + +def is_malformed_schema_error(exc: BaseException) -> bool: + """Only SQLite's explicit malformed-schema text. A generic "disk image is + malformed" (SQLITE_CORRUPT) may be any B-tree/freelist page and does not + prove canonical rows intact, so runtime repair must fail closed on it.""" + return isinstance(exc, sqlite3.DatabaseError) and any( + marker in str(exc).lower() for marker in _MALFORMED_SCHEMA_MARKERS + ) + + +# "Filesystem cannot accept another write" substrings (OSError, sqlite3, and +# wrapped RPC strings all match the same helper). +_DISK_FULL_MARKERS = ( + "no space left on device", + "not enough space", + "database or disk is full", # SQLITE_FULL + "disk full", + "full disk", + "enospc", +) + + +def is_disk_full_error(exc: BaseException | str | None) -> bool: + """Disk-full / ENOSPC: OSError(ENOSPC), SQLITE_FULL, or matching strings.""" + if exc is None: + return False + if isinstance(exc, OSError) and getattr(exc, "errno", None) == errno.ENOSPC: + return True + lowered = (exc if isinstance(exc, str) else str(exc)).lower() + return any(marker in lowered for marker in _DISK_FULL_MARKERS) + + +# Every classify_persistence_error bucket; consumers enumerate this tuple so a +# new bucket can never silently desynchronize them. +PERSISTENCE_ERROR_CAUSES = ( + "locked", "compression", "compression_closed", "turn_lease", "corrupt", "replaced", "disk", + "unknown", +) + + +# "Database FILE structurally damaged" substrings. NOTE: "database disk image is +# malformed" contains "disk", so this check MUST run before the disk bucket in +# classify_persistence_error or B-tree corruption reads as "free some disk space". +_DB_CORRUPTION_MARKERS = ( + "malformed", # "database disk image is malformed" (SQLITE_CORRUPT) + "file is not a database", # SQLITE_NOTADB (also connection-level poisoning) + "not a database", + "database corruption", +) + + +def classify_persistence_error(exc_or_str) -> str: + """Coarse cause bucket (PERSISTENCE_ERROR_CAUSES) so the user's guidance + matches: "locked" = busy, retry; "disk" = full/read-only/permissions; + "compression" = a live lease refused the write; "compression_closed" = adopt + the rotated session id; "turn_lease" = fencing, not storage; "corrupt" = + file damage (repair path, not disk space); "replaced" = stop writing.""" + if exc_or_str is None: + return "unknown" + # Lease refusals contain neither "locked" nor "busy": match by type, then by + # phrase for strings that survived RPC wrapping. + if isinstance(exc_or_str, SessionTurnLeaseLostError): + return "turn_lease" + if isinstance(exc_or_str, CompressionSessionClosedError): + return "compression_closed" + if isinstance(exc_or_str, CompressionSessionBusyError): + return "compression" + if isinstance(exc_or_str, StateDbReplacedError): # incl. DeletedWalGenerationError + return "replaced" + if isinstance(exc_or_str, StateDbCorruptError): + return "corrupt" + text = str(exc_or_str).lower() + if "turn lease" in text: + return "turn_lease" + if "closed by compression" in text: + return "compression_closed" + if "being compressed" in text or "compression lease" in text: + return "compression" + if "was replaced underneath" in text: + return "replaced" + if "deleted state.db-wal" in text or "deleted state.db-shm" in text: + return "replaced" + # Corruption BEFORE the lock/disk buckets: "disk image is malformed" + # contains "disk" and some wrapped strings mention "locked" recovery. + if any(marker in text for marker in _DB_CORRUPTION_MARKERS): + return "corrupt" + if "locked" in text or "busy" in text: + return "locked" + if is_disk_full_error(exc_or_str) or "disk" in text or "readonly" in text or "read-only" in text: + return "disk" + return "unknown" + + +class CompressionSessionClosedError(RuntimeError): + """A durable write targeted a parent already closed by compression.""" + + def __init__(self, session_id: str): + self.session_id = session_id + super().__init__( + f"Session {session_id!r} is closed by compression; " + "adopt its live continuation before appending messages" + ) + + +class CompressionSessionBusyError(RuntimeError): + """A non-owner tried to write while compression owns the session.""" + + +class SessionCompressionInProgressError(CompressionSessionBusyError): + """A concurrent writer collided with a *live* compression lock — transient + (the compressor publishes in seconds; ``_execute_write`` waits), unlike the + parent class's other case (a compressor whose own lease is gone: permanent, + fail fast). Subclassing keeps every existing handler working.""" + + +class SessionTurnLeaseLostError(RuntimeError): + """A transcript write presented a turn-lease holder that no longer owns it. + Fail-fast fencing (no ``_execute_write`` retry): a later writer may already + be persisting a newer turn, and landing this one would interleave a stale reply.""" + + +class StateDbReplacedError(RuntimeError): + """The state.db path no longer names the file this SessionDB opened + (out-of-band cp/mv/restore). In-place FTS repair and fail-open trigger + dropping cannot fix a generation mismatch; they amplify it.""" + + +class DeletedWalGenerationError(StateDbReplacedError): + """A live process holds a deleted state.db-wal / -shm generation. Opening or + writing through this handle would mint a second WAL inode (split-brain -> + intermittent SQLITE_CORRUPT / IOERR). Stop the writers; never unlink the WAL + yourself. Subclasses StateDbReplacedError so every consumer that diverts + transcripts on a replaced store handles this identically.""" + + +# SQLite header application_id (offset 68). Distinct from inode: ``cp`` onto the +# same path keeps st_ino and truncates+rewrites. +_STATE_DB_APPLICATION_ID_OFFSET = 68 +_STATE_DB_GENERATION_KEY = "db_file_generation" +_STATE_DB_REPLACED_MSG = ( + "FATAL: state.db was replaced underneath the gateway; refusing further " + "writes to this file. Divert transcripts to sessions/.jsonl (and the " + "gateway pending_messages spool) and restore or reopen after operator intervention." +) +_DELETED_WAL_GENERATION_MSG = ( + "FATAL: a live process holds a deleted state.db-wal or state.db-shm " + "inode while the path names a different (or missing) generation. " + "Refusing to open or write so a second WAL cannot be minted. " + "Stop the gateway, dashboard, and cron writers that hold the deleted " + "sidecar, then reopen. Do not delete the WAL yourself. " + "database.journal_mode: delete is operator containment, not a new default." +) + + +class StateDbCorruptError(sqlite3.DatabaseError): + """A live SessionDB observed structural (non-FTS, non-replaced) corruption and + is quarantined: sticky for the handle's life — writes fail fast, no reopen, + no close-time checkpoint (a handle that kept writing after the first error + checkpointed 15 pages under wrong page numbers and turned a readable file + into "file is not a database"; SQLITE_DBCONFIG_NO_CKPT_ON_CLOSE on 3.12+ + also stops SQLite's own). Subclasses sqlite3.DatabaseError so every degrade + path keeps working. Recovery boundary: restart on a repaired/restored file.""" + + +_STATE_DB_CORRUPT_MSG = ( + "FATAL: state.db reported structural corruption (database disk image is " + "malformed outside the FTS shadow tables) on a live handle; refusing further " + "writes, automatic reopen, and the close-time WAL checkpoint on this file. " + "Stop the gateway, then run `hermes sessions recover --source " + "--inspect-only` or restore a snapshot. Unwritten transcripts are diverted to " + "sessions/.jsonl (and the gateway pending_messages spool)." +) diff --git a/hermes_state_guard.py b/hermes_state_guard.py new file mode 100644 index 0000000000..e9506dfc5a --- /dev/null +++ b/hermes_state_guard.py @@ -0,0 +1,148 @@ +"""Live-DB test-isolation guard and the per-process "last init error" record. +Every SessionDB construction resolves its path through _ensure_test_isolation +so a pytest-context process (env OR ancestry) can never open a production +state.db; env-based so subprocess children are protected too.""" + +import os +import sys +import threading +from pathlib import Path +from typing import Any, Optional + +try: # Hard dependency, but tolerate scaffold-phase imports before pip install. + import psutil +except ImportError: # pragma: no cover - stripped/scaffold installs only + psutil = None # type: ignore[assignment] + +# Field evidence: pytest fixture rows landed in the production state.db and a +# pytest-spawned child flipped the journal mode under the live WAL writer, +# destroying committed transcripts; any HERMES_HOME escape (fixture ordering, a +# child spawned without it, a shell exporting the real home) fell through silently. + +#: Env twin of ``_STATE_DB_GUARD_BYPASS`` for child processes (a module global +#: cannot cross a process boundary, and ancestry arms the guard there). +_STATE_DB_GUARD_BYPASS_ENV = "HERMES_STATE_DB_GUARD_BYPASS" + + +def _real_platform_state_root() -> Optional[Path]: + """The REAL platform-default Hermes root. Avoids ``Path.home()`` / + ``hermes_constants``: tests monkeypatch Path.home to a tempdir while this + module is imported lazily, which would misidentify the hermetic home as + production or miss the real one. ``expanduser`` reads HOME/passwd, which the + conftest never rewrites.""" + try: + if sys.platform == "win32": + base = os.environ.get("LOCALAPPDATA", "").strip() + root = ( + Path(base) / "hermes" + if base + else Path(os.path.expanduser("~")) / "AppData" / "Local" / "hermes" + ) + else: + root = Path(os.path.expanduser("~")) / ".hermes" + return root.resolve() + except Exception: + return None + + +#: Exported by the hermetic conftest alongside the HERMES_HOME redirect (value: +#: the isolation root). Unlike PYTEST_* (scrubbed by tests that rebuild a child +#: env) it is OURS and inherits by default, so a child carrying it that resolves +#: a production DB is by definition an isolation escape. +_TEST_ISOLATION_MARKER_ENV = "HERMES_TEST_ISOLATION" + + +def _running_under_pytest() -> bool: + """True when this process (or a parent test process) is a pytest run.""" + return bool( + os.environ.get("PYTEST_CURRENT_TEST") + or os.environ.get("PYTEST_VERSION") + or os.environ.get(_TEST_ISOLATION_MARKER_ENV) + ) + + +#: pytest launcher names, matched against each argv token's *basename* so +#: ``/tmp/pytest-of-dev/...`` paths cannot false-positive. +_PYTEST_LAUNCHER_NAMES = frozenset({"pytest", "py.test", "pytest.exe", "py.test.exe"}) + +#: Memoised ancestry answer: the tree above us doesn't change; keep the hot path free. +_PYTEST_ANCESTOR: Optional[bool] = None + + +def _process_looks_like_pytest(proc: Any) -> bool: + """True when *proc*'s command line is a pytest invocation (``pytest ...`` or + ``python -m pytest``). Unreadable cmdline => not pytest: guessing the other + way would refuse production opens for unrelated reasons.""" + try: + cmdline = proc.cmdline() or [] + except Exception: + return False + for arg in cmdline: + try: + # Split on both separators on every host: os.path.basename is + # POSIX-only under Linux and would leave a Windows-style path + # intact, making the matcher's answer depend on the platform. + name = str(arg).strip('"').strip("'").replace("\\", "/").rsplit("/", 1)[-1].lower() + except Exception: + continue + if name in _PYTEST_LAUNCHER_NAMES: + return True + return False + + +def _has_pytest_ancestor() -> bool: + """True when an ancestor process is a pytest run. A child spawned with a + rebuilt env loses PYTEST_* and the HERMES_HOME redirect together — aiming at + production AND disarming the guard in one step; ancestry survives that. + Fails open without psutil / on walk errors (never block real user runs).""" + global _PYTEST_ANCESTOR + if _PYTEST_ANCESTOR is not None: + return _PYTEST_ANCESTOR + found = False + if psutil is not None: + try: + found = any(_process_looks_like_pytest(p) for p in psutil.Process().parents()) + except Exception: + found = False + _PYTEST_ANCESTOR = found + return found + + +def _in_test_context() -> bool: + """Test run by environment or ancestry. Env first (two dict lookups); the + memoised ancestry walk runs at most once per real ``hermes`` invocation.""" + return _running_under_pytest() or _has_pytest_ancestor() + + +def _is_production_state_db(resolved: Path, root: Path) -> bool: + """*resolved* is ``/state.db`` or ``/profiles//state.db``. + Deeper scratch paths (repo worktrees under ~/.hermes/hermes-agent/...) are + deliberately NOT matched so hermetic tests cannot false-positive.""" + if resolved.parent == root: + return True + try: + parts = resolved.relative_to(root).parts + except ValueError: + return False + return len(parts) == 3 and parts[0] == "profiles" + + +# Last SessionDB() init error, per-process; surfaced by /resume-style slash +# commands so users know WHY. Only SessionDB.__init__ writes it (kanban_db +# failures are reported via their own callers, by design). +_last_init_error: Optional[str] = None +_last_init_error_lock = threading.Lock() + + +def _set_last_init_error(msg: Optional[str]) -> None: + """Record (or clear with None) the most recent state.db init failure. + __init__ only SETs on failure and never clears on success: a concurrent + successful open would erase the cause another thread's /resume is about to format.""" + global _last_init_error + with _last_init_error_lock: + _last_init_error = msg + + +def get_last_init_error() -> Optional[str]: + """Most recent state.db init failure (None if none/never attempted).""" + return _last_init_error