refactor(state): extract error types/classifiers to hermes_state_errors and the live-DB guard to hermes_state_guard
This commit is contained in:
387
hermes_state.py
387
hermes_state.py
@@ -45,6 +45,21 @@ from hermes_state_common import ( # noqa: F401 (re-exported; tests import from
|
||||
FTS_TRIGRAM_SQL, LEGACY_FTS_SQL, LEGACY_FTS_TRIGRAM_SQL, SCHEMA_SQL, SCHEMA_VERSION,
|
||||
stat_db_file_identity as _stat_db_file_identity,
|
||||
)
|
||||
from hermes_state_errors import ( # noqa: F401 (re-exported; the historical import path)
|
||||
_DB_CORRUPTION_MARKERS, _DELETED_WAL_GENERATION_MSG, _DISK_FULL_MARKERS, _DISK_IO_ERROR_MARKER,
|
||||
_MALFORMED_DB_MARKERS, _MALFORMED_SCHEMA_MARKERS, _STATE_DB_APPLICATION_ID_OFFSET,
|
||||
_STATE_DB_CORRUPT_MSG, _STATE_DB_GENERATION_KEY, _STATE_DB_REPLACED_MSG,
|
||||
_TRANSIENT_SQLITE_MARKERS, PERSISTENCE_ERROR_CAUSES, CompressionSessionBusyError,
|
||||
CompressionSessionClosedError, DeletedWalGenerationError, SessionCompressionInProgressError,
|
||||
SessionTurnLeaseLostError, StateDbCorruptError, StateDbReplacedError, _is_no_more_rows,
|
||||
classify_persistence_error, is_disk_full_error, is_malformed_db_error,
|
||||
is_malformed_schema_error, is_transient_sqlite_error,
|
||||
)
|
||||
from hermes_state_guard import ( # noqa: F401 (re-exported; tests patch hermes_state.<name>)
|
||||
_PYTEST_LAUNCHER_NAMES, _STATE_DB_GUARD_BYPASS_ENV, _TEST_ISOLATION_MARKER_ENV,
|
||||
_has_pytest_ancestor, _in_test_context, _is_production_state_db, _process_looks_like_pytest,
|
||||
_real_platform_state_root, _running_under_pytest, _set_last_init_error, get_last_init_error,
|
||||
)
|
||||
from hermes_state_portability import SessionPortabilityMixin
|
||||
from hermes_state_telegram import SessionTelegramTopicsMixin
|
||||
from hermes_state_schema import SessionSchemaMixin
|
||||
@@ -361,6 +376,12 @@ _READ_OPEN_RETRY_SECONDS = 60.0
|
||||
_READ_ONLY_IOERR_RETRY_ATTEMPTS = 3
|
||||
_READ_ONLY_IOERR_RETRY_BACKOFF_S = 0.05
|
||||
|
||||
def _is_transient_read_only_ioerr(exc: sqlite3.OperationalError, *, attempt: int) -> bool:
|
||||
"""Retry a read-only open? See _READ_ONLY_IOERR_RETRY_ATTEMPTS: a
|
||||
persistent IOERR still exhausts the budget and propagates."""
|
||||
return attempt < _READ_ONLY_IOERR_RETRY_ATTEMPTS and _DISK_IO_ERROR_MARKER in str(exc).lower()
|
||||
|
||||
|
||||
# Ceiling on read-only connections ALIVE at once against one database FILE
|
||||
# (idle pooled + checked out, summed over every SessionDB on that file). One
|
||||
# constant for both the pool maxsize and the permit count: a LifoQueue only caps
|
||||
@@ -583,120 +604,17 @@ def _default_db_path() -> Path:
|
||||
return get_hermes_home() / "state.db"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Live-DB test-isolation guard
|
||||
# ---------------------------------------------------------------------------
|
||||
# Field evidence: pytest fixture rows landed in the production state.db and a
|
||||
# pytest-spawned child flipped the journal mode under the live WAL writer,
|
||||
# destroying committed transcripts. The hermetic conftest redirects HERMES_HOME
|
||||
# per test, but any escape (fixture ordering, a child spawned without
|
||||
# HERMES_HOME, a shell exporting the real home) fell through silently. EVERY
|
||||
# SessionDB construction resolves its path here, so under pytest a production
|
||||
# state.db fails hard. Env-based, so subprocess children are protected too.
|
||||
|
||||
# Live-DB guard knobs live HERE (not in hermes_state_guard): the hermetic conftest
|
||||
# monkeypatches ``hermes_state._STATE_DB_GUARD_BYPASS`` / ``_EXTRA_DENY_ROOTS``.
|
||||
#: Escape hatch for tests that genuinely need the real DB (conftest sets it for
|
||||
#: ``@pytest.mark.live_system_guard_bypass``); scripts may set it explicitly.
|
||||
_STATE_DB_GUARD_BYPASS = False
|
||||
|
||||
#: Env twin of ``_STATE_DB_GUARD_BYPASS`` for child processes (a module global
|
||||
#: cannot cross a process boundary, and ancestry arms the guard there).
|
||||
_STATE_DB_GUARD_BYPASS_ENV = "HERMES_STATE_DB_GUARD_BYPASS"
|
||||
|
||||
#: Extra production roots to refuse; conftest injects the pre-sandbox root so
|
||||
#: custom-HERMES_HOME deployments are covered too.
|
||||
_STATE_DB_GUARD_EXTRA_DENY_ROOTS: Tuple[Path, ...] = ()
|
||||
|
||||
|
||||
def _real_platform_state_root() -> Optional[Path]:
|
||||
"""The REAL platform-default Hermes root. Avoids ``Path.home()`` /
|
||||
``hermes_constants``: tests monkeypatch Path.home to a tempdir while this
|
||||
module is imported lazily, which would misidentify the hermetic home as
|
||||
production or miss the real one. ``expanduser`` reads HOME/passwd, which the
|
||||
conftest never rewrites."""
|
||||
try:
|
||||
if sys.platform == "win32":
|
||||
base = os.environ.get("LOCALAPPDATA", "").strip()
|
||||
root = (
|
||||
Path(base) / "hermes"
|
||||
if base
|
||||
else Path(os.path.expanduser("~")) / "AppData" / "Local" / "hermes"
|
||||
)
|
||||
else:
|
||||
root = Path(os.path.expanduser("~")) / ".hermes"
|
||||
return root.resolve()
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
#: Exported by the hermetic conftest alongside the HERMES_HOME redirect (value:
|
||||
#: the isolation root). Unlike PYTEST_* (scrubbed by tests that rebuild a child
|
||||
#: env) it is OURS and inherits by default, so a child carrying it that resolves
|
||||
#: a production DB is by definition an isolation escape.
|
||||
_TEST_ISOLATION_MARKER_ENV = "HERMES_TEST_ISOLATION"
|
||||
|
||||
|
||||
def _running_under_pytest() -> bool:
|
||||
"""True when this process (or a parent test process) is a pytest run."""
|
||||
return bool(
|
||||
os.environ.get("PYTEST_CURRENT_TEST")
|
||||
or os.environ.get("PYTEST_VERSION")
|
||||
or os.environ.get(_TEST_ISOLATION_MARKER_ENV)
|
||||
)
|
||||
|
||||
|
||||
#: pytest launcher names, matched against each argv token's *basename* so
|
||||
#: ``/tmp/pytest-of-dev/...`` paths cannot false-positive.
|
||||
_PYTEST_LAUNCHER_NAMES = frozenset({"pytest", "py.test", "pytest.exe", "py.test.exe"})
|
||||
|
||||
#: Memoised ancestry answer: the tree above us doesn't change; keep the hot path free.
|
||||
_PYTEST_ANCESTOR: Optional[bool] = None
|
||||
|
||||
|
||||
def _process_looks_like_pytest(proc: Any) -> bool:
|
||||
"""True when *proc*'s command line is a pytest invocation (``pytest ...`` or
|
||||
``python -m pytest``). Unreadable cmdline => not pytest: guessing the other
|
||||
way would refuse production opens for unrelated reasons."""
|
||||
try:
|
||||
cmdline = proc.cmdline() or []
|
||||
except Exception:
|
||||
return False
|
||||
for arg in cmdline:
|
||||
try:
|
||||
# Split on both separators on every host: os.path.basename is
|
||||
# POSIX-only under Linux and would leave a Windows-style path
|
||||
# intact, making the matcher's answer depend on the platform.
|
||||
name = str(arg).strip('"').strip("'").replace("\\", "/").rsplit("/", 1)[-1].lower()
|
||||
except Exception:
|
||||
continue
|
||||
if name in _PYTEST_LAUNCHER_NAMES:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _has_pytest_ancestor() -> bool:
|
||||
"""True when an ancestor process is a pytest run. A child spawned with a
|
||||
rebuilt env loses PYTEST_* and the HERMES_HOME redirect together — aiming at
|
||||
production AND disarming the guard in one step; ancestry survives that.
|
||||
Fails open without psutil / on walk errors (never block real user runs)."""
|
||||
global _PYTEST_ANCESTOR
|
||||
if _PYTEST_ANCESTOR is not None:
|
||||
return _PYTEST_ANCESTOR
|
||||
found = False
|
||||
if psutil is not None:
|
||||
try:
|
||||
found = any(_process_looks_like_pytest(p) for p in psutil.Process().parents())
|
||||
except Exception:
|
||||
found = False
|
||||
_PYTEST_ANCESTOR = found
|
||||
return found
|
||||
|
||||
|
||||
def _in_test_context() -> bool:
|
||||
"""Test run by environment or ancestry. Env first (two dict lookups); the
|
||||
memoised ancestry walk runs at most once per real ``hermes`` invocation."""
|
||||
return _running_under_pytest() or _has_pytest_ancestor()
|
||||
|
||||
|
||||
def _production_state_roots() -> List[Path]:
|
||||
roots: List[Path] = []
|
||||
real_root = _real_platform_state_root()
|
||||
@@ -710,19 +628,6 @@ def _production_state_roots() -> List[Path]:
|
||||
return roots
|
||||
|
||||
|
||||
def _is_production_state_db(resolved: Path, root: Path) -> bool:
|
||||
"""*resolved* is ``<root>/state.db`` or ``<root>/profiles/<name>/state.db``.
|
||||
Deeper scratch paths (repo worktrees under ~/.hermes/hermes-agent/...) are
|
||||
deliberately NOT matched so hermetic tests cannot false-positive."""
|
||||
if resolved.parent == root:
|
||||
return True
|
||||
try:
|
||||
parts = resolved.relative_to(root).parts
|
||||
except ValueError:
|
||||
return False
|
||||
return len(parts) == 3 and parts[0] == "profiles"
|
||||
|
||||
|
||||
def _ensure_test_isolation(db_path: Path) -> None:
|
||||
"""Raise RuntimeError before any connection/mkdir/pragma/byte probe when a
|
||||
pytest-context process (env OR ancestry, see :func:`_in_test_context`)
|
||||
@@ -749,27 +654,6 @@ def _ensure_test_isolation(db_path: Path) -> None:
|
||||
)
|
||||
|
||||
|
||||
# Last SessionDB() init error, per-process; surfaced by /resume-style slash
|
||||
# commands so users know WHY. Only SessionDB.__init__ writes it (kanban_db
|
||||
# failures are reported via their own callers, by design).
|
||||
_last_init_error: Optional[str] = None
|
||||
_last_init_error_lock = threading.Lock()
|
||||
|
||||
|
||||
def _set_last_init_error(msg: Optional[str]) -> None:
|
||||
"""Record (or clear with None) the most recent state.db init failure.
|
||||
__init__ only SETs on failure and never clears on success: a concurrent
|
||||
successful open would erase the cause another thread's /resume is about to format."""
|
||||
global _last_init_error
|
||||
with _last_init_error_lock:
|
||||
_last_init_error = msg
|
||||
|
||||
|
||||
def get_last_init_error() -> Optional[str]:
|
||||
"""Most recent state.db init failure (None if none/never attempted)."""
|
||||
return _last_init_error
|
||||
|
||||
|
||||
# Openings of the background-review harness prompts (agent/background_review.py),
|
||||
# matched case-sensitively against leading user/system content.
|
||||
_REVIEW_HARNESS_PREFIXES = (
|
||||
@@ -848,156 +732,12 @@ def format_session_db_unavailable(prefix: str = "Session database not available"
|
||||
return f"{prefix}: {cause}{hint}."
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Malformed-schema recovery: ``sqlite_master`` itself is inconsistent (typically
|
||||
# a DUPLICATE ``CREATE VIRTUAL TABLE messages_fts`` row). SQLite parses the
|
||||
# whole schema while preparing the FIRST statement, so EVERY statement raises —
|
||||
# including ``PRAGMA journal_mode`` (it trips in apply_wal_with_fallback during
|
||||
# __init__, before _init_schema) and plain ``DROP TABLE``; only
|
||||
# ``PRAGMA writable_schema=ON`` + sqlite_master surgery still work. Canonical
|
||||
# sessions/messages are intact; recovery rebuilds only the FTS layer.
|
||||
_MALFORMED_SCHEMA_MARKERS = ("malformed database schema",)
|
||||
_MALFORMED_DB_MARKERS = (*_MALFORMED_SCHEMA_MARKERS, "database disk image is malformed")
|
||||
|
||||
# Auto-repair at most once per DB path per process (no repair loops; serialises
|
||||
# concurrent web_server / gateway opens on the same malformed file).
|
||||
_repair_attempted_paths: set[str] = set()
|
||||
_repair_attempt_lock = threading.Lock()
|
||||
|
||||
|
||||
def is_malformed_db_error(exc: BaseException) -> bool:
|
||||
"""Malformed-schema OR generic corrupt-image error. Diagnostics / offline
|
||||
recovery only — runtime repair must use :func:`is_malformed_schema_error`."""
|
||||
return isinstance(exc, sqlite3.DatabaseError) and any(
|
||||
marker in str(exc).lower() for marker in _MALFORMED_DB_MARKERS
|
||||
)
|
||||
|
||||
|
||||
# SQLITE_IOERR as a substring (wrapped strings still classify); shared by the
|
||||
# read-only open retry and the write-path BEGIN retry.
|
||||
_DISK_IO_ERROR_MARKER = "disk i/o error"
|
||||
|
||||
# "Store BUSY, not gone" — HTTP callers map these to 503 instead of 500.
|
||||
# Corruption deliberately absent: a malformed store must surface, not be
|
||||
# retried into a timeout.
|
||||
_TRANSIENT_SQLITE_MARKERS = (
|
||||
_DISK_IO_ERROR_MARKER, "database is locked", "database table is locked", "busy",
|
||||
)
|
||||
|
||||
|
||||
def _is_no_more_rows(exc: sqlite3.Error) -> bool:
|
||||
"""Transient engine error on contended WAL appends; the identical write succeeds
|
||||
standalone, so it retries like locked/busy. Message-scoped because some builds
|
||||
raise it as InterfaceError (outside DatabaseError)."""
|
||||
return "no more rows available" in str(exc).lower()
|
||||
|
||||
|
||||
def is_transient_sqlite_error(exc: BaseException) -> bool:
|
||||
""""Busy right now", not "damaged". One predicate so the read-only open
|
||||
retry and the HTTP 503-vs-500 split cannot drift apart."""
|
||||
return isinstance(exc, sqlite3.OperationalError) and any(
|
||||
marker in str(exc).lower() for marker in _TRANSIENT_SQLITE_MARKERS
|
||||
)
|
||||
|
||||
|
||||
def _is_transient_read_only_ioerr(exc: sqlite3.OperationalError, *, attempt: int) -> bool:
|
||||
"""Retry a read-only open? See _READ_ONLY_IOERR_RETRY_ATTEMPTS: a
|
||||
persistent IOERR still exhausts the budget and propagates."""
|
||||
return attempt < _READ_ONLY_IOERR_RETRY_ATTEMPTS and _DISK_IO_ERROR_MARKER in str(exc).lower()
|
||||
|
||||
|
||||
def is_malformed_schema_error(exc: BaseException) -> bool:
|
||||
"""Only SQLite's explicit malformed-schema text. A generic "disk image is
|
||||
malformed" (SQLITE_CORRUPT) may be any B-tree/freelist page and does not
|
||||
prove canonical rows intact, so runtime repair must fail closed on it."""
|
||||
return isinstance(exc, sqlite3.DatabaseError) and any(
|
||||
marker in str(exc).lower() for marker in _MALFORMED_SCHEMA_MARKERS
|
||||
)
|
||||
|
||||
|
||||
# "Filesystem cannot accept another write" substrings (OSError, sqlite3, and
|
||||
# wrapped RPC strings all match the same helper).
|
||||
_DISK_FULL_MARKERS = (
|
||||
"no space left on device",
|
||||
"not enough space",
|
||||
"database or disk is full", # SQLITE_FULL
|
||||
"disk full",
|
||||
"full disk",
|
||||
"enospc",
|
||||
)
|
||||
|
||||
|
||||
def is_disk_full_error(exc: BaseException | str | None) -> bool:
|
||||
"""Disk-full / ENOSPC: OSError(ENOSPC), SQLITE_FULL, or matching strings."""
|
||||
if exc is None:
|
||||
return False
|
||||
if isinstance(exc, OSError) and getattr(exc, "errno", None) == errno.ENOSPC:
|
||||
return True
|
||||
lowered = (exc if isinstance(exc, str) else str(exc)).lower()
|
||||
return any(marker in lowered for marker in _DISK_FULL_MARKERS)
|
||||
|
||||
|
||||
# Every classify_persistence_error bucket; consumers enumerate this tuple so a
|
||||
# new bucket can never silently desynchronize them.
|
||||
PERSISTENCE_ERROR_CAUSES = (
|
||||
"locked", "compression", "compression_closed", "turn_lease", "corrupt", "replaced", "disk",
|
||||
"unknown",
|
||||
)
|
||||
|
||||
|
||||
# "Database FILE structurally damaged" substrings. NOTE: "database disk image is
|
||||
# malformed" contains "disk", so this check MUST run before the disk bucket in
|
||||
# classify_persistence_error or B-tree corruption reads as "free some disk space".
|
||||
_DB_CORRUPTION_MARKERS = (
|
||||
"malformed", # "database disk image is malformed" (SQLITE_CORRUPT)
|
||||
"file is not a database", # SQLITE_NOTADB (also connection-level poisoning)
|
||||
"not a database",
|
||||
"database corruption",
|
||||
)
|
||||
|
||||
|
||||
def classify_persistence_error(exc_or_str) -> str:
|
||||
"""Coarse cause bucket (PERSISTENCE_ERROR_CAUSES) so the user's guidance
|
||||
matches: "locked" = busy, retry; "disk" = full/read-only/permissions;
|
||||
"compression" = a live lease refused the write; "compression_closed" = adopt
|
||||
the rotated session id; "turn_lease" = fencing, not storage; "corrupt" =
|
||||
file damage (repair path, not disk space); "replaced" = stop writing."""
|
||||
if exc_or_str is None:
|
||||
return "unknown"
|
||||
# Lease refusals contain neither "locked" nor "busy": match by type, then by
|
||||
# phrase for strings that survived RPC wrapping.
|
||||
if isinstance(exc_or_str, SessionTurnLeaseLostError):
|
||||
return "turn_lease"
|
||||
if isinstance(exc_or_str, CompressionSessionClosedError):
|
||||
return "compression_closed"
|
||||
if isinstance(exc_or_str, CompressionSessionBusyError):
|
||||
return "compression"
|
||||
if isinstance(exc_or_str, StateDbReplacedError): # incl. DeletedWalGenerationError
|
||||
return "replaced"
|
||||
if isinstance(exc_or_str, StateDbCorruptError):
|
||||
return "corrupt"
|
||||
text = str(exc_or_str).lower()
|
||||
if "turn lease" in text:
|
||||
return "turn_lease"
|
||||
if "closed by compression" in text:
|
||||
return "compression_closed"
|
||||
if "being compressed" in text or "compression lease" in text:
|
||||
return "compression"
|
||||
if "was replaced underneath" in text:
|
||||
return "replaced"
|
||||
if "deleted state.db-wal" in text or "deleted state.db-shm" in text:
|
||||
return "replaced"
|
||||
# Corruption BEFORE the lock/disk buckets: "disk image is malformed"
|
||||
# contains "disk" and some wrapped strings mention "locked" recovery.
|
||||
if any(marker in text for marker in _DB_CORRUPTION_MARKERS):
|
||||
return "corrupt"
|
||||
if "locked" in text or "busy" in text:
|
||||
return "locked"
|
||||
if is_disk_full_error(exc_or_str) or "disk" in text or "readonly" in text or "read-only" in text:
|
||||
return "disk"
|
||||
return "unknown"
|
||||
|
||||
|
||||
# Cross-process schema-surgery lock: ``_repair_attempt_lock`` covers one
|
||||
# interpreter only, while gateway, Desktop backend, CLI and TUI worker share the
|
||||
# file and each used to run surgery + VACUUM on top of the winner's. Timeout
|
||||
@@ -1123,87 +863,6 @@ def load_fts5_cjk_extension(conn: sqlite3.Connection) -> bool:
|
||||
return False
|
||||
|
||||
|
||||
class CompressionSessionClosedError(RuntimeError):
|
||||
"""A durable write targeted a parent already closed by compression."""
|
||||
|
||||
def __init__(self, session_id: str):
|
||||
self.session_id = session_id
|
||||
super().__init__(
|
||||
f"Session {session_id!r} is closed by compression; "
|
||||
"adopt its live continuation before appending messages"
|
||||
)
|
||||
|
||||
|
||||
class CompressionSessionBusyError(RuntimeError):
|
||||
"""A non-owner tried to write while compression owns the session."""
|
||||
|
||||
|
||||
class SessionCompressionInProgressError(CompressionSessionBusyError):
|
||||
"""A concurrent writer collided with a *live* compression lock — transient
|
||||
(the compressor publishes in seconds; ``_execute_write`` waits), unlike the
|
||||
parent class's other case (a compressor whose own lease is gone: permanent,
|
||||
fail fast). Subclassing keeps every existing handler working."""
|
||||
|
||||
|
||||
class SessionTurnLeaseLostError(RuntimeError):
|
||||
"""A transcript write presented a turn-lease holder that no longer owns it.
|
||||
Fail-fast fencing (no ``_execute_write`` retry): a later writer may already
|
||||
be persisting a newer turn, and landing this one would interleave a stale reply."""
|
||||
|
||||
|
||||
class StateDbReplacedError(RuntimeError):
|
||||
"""The state.db path no longer names the file this SessionDB opened
|
||||
(out-of-band cp/mv/restore). In-place FTS repair and fail-open trigger
|
||||
dropping cannot fix a generation mismatch; they amplify it."""
|
||||
|
||||
|
||||
class DeletedWalGenerationError(StateDbReplacedError):
|
||||
"""A live process holds a deleted state.db-wal / -shm generation. Opening or
|
||||
writing through this handle would mint a second WAL inode (split-brain ->
|
||||
intermittent SQLITE_CORRUPT / IOERR). Stop the writers; never unlink the WAL
|
||||
yourself. Subclasses StateDbReplacedError so every consumer that diverts
|
||||
transcripts on a replaced store handles this identically."""
|
||||
|
||||
|
||||
# SQLite header application_id (offset 68). Distinct from inode: ``cp`` onto the
|
||||
# same path keeps st_ino and truncates+rewrites.
|
||||
_STATE_DB_APPLICATION_ID_OFFSET = 68
|
||||
_STATE_DB_GENERATION_KEY = "db_file_generation"
|
||||
_STATE_DB_REPLACED_MSG = (
|
||||
"FATAL: state.db was replaced underneath the gateway; refusing further "
|
||||
"writes to this file. Divert transcripts to sessions/<id>.jsonl (and the "
|
||||
"gateway pending_messages spool) and restore or reopen after operator intervention."
|
||||
)
|
||||
_DELETED_WAL_GENERATION_MSG = (
|
||||
"FATAL: a live process holds a deleted state.db-wal or state.db-shm "
|
||||
"inode while the path names a different (or missing) generation. "
|
||||
"Refusing to open or write so a second WAL cannot be minted. "
|
||||
"Stop the gateway, dashboard, and cron writers that hold the deleted "
|
||||
"sidecar, then reopen. Do not delete the WAL yourself. "
|
||||
"database.journal_mode: delete is operator containment, not a new default."
|
||||
)
|
||||
|
||||
|
||||
class StateDbCorruptError(sqlite3.DatabaseError):
|
||||
"""A live SessionDB observed structural (non-FTS, non-replaced) corruption and
|
||||
is quarantined: sticky for the handle's life — writes fail fast, no reopen,
|
||||
no close-time checkpoint (a handle that kept writing after the first error
|
||||
checkpointed 15 pages under wrong page numbers and turned a readable file
|
||||
into "file is not a database"; SQLITE_DBCONFIG_NO_CKPT_ON_CLOSE on 3.12+
|
||||
also stops SQLite's own). Subclasses sqlite3.DatabaseError so every degrade
|
||||
path keeps working. Recovery boundary: restart on a repaired/restored file."""
|
||||
|
||||
|
||||
_STATE_DB_CORRUPT_MSG = (
|
||||
"FATAL: state.db reported structural corruption (database disk image is "
|
||||
"malformed outside the FTS shadow tables) on a live handle; refusing further "
|
||||
"writes, automatic reopen, and the close-time WAL checkpoint on this file. "
|
||||
"Stop the gateway, then run `hermes sessions recover --source <state.db> "
|
||||
"--inspect-only` or restore a snapshot. Unwritten transcripts are diverted to "
|
||||
"sessions/<id>.jsonl (and the gateway pending_messages spool)."
|
||||
)
|
||||
|
||||
|
||||
def divert_session_transcript_jsonl(session_id: str, messages) -> "Optional[Path]":
|
||||
"""Append pending messages to HERMES_HOME/sessions/<id>.jsonl (state.db was
|
||||
replaced under a live process). Returns the path, or None if nothing to write."""
|
||||
|
||||
225
hermes_state_errors.py
Normal file
225
hermes_state_errors.py
Normal file
@@ -0,0 +1,225 @@
|
||||
"""Exception types and error-classification predicates for the state store.
|
||||
Shared by hermes_state and its mixins; string predicates match wrapped RPC
|
||||
strings as well as live sqlite3 exceptions."""
|
||||
|
||||
import errno
|
||||
import sqlite3
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Malformed-schema recovery: ``sqlite_master`` itself is inconsistent (typically
|
||||
# a DUPLICATE ``CREATE VIRTUAL TABLE messages_fts`` row). SQLite parses the
|
||||
# whole schema while preparing the FIRST statement, so EVERY statement raises —
|
||||
# including ``PRAGMA journal_mode`` (it trips in apply_wal_with_fallback during
|
||||
# __init__, before _init_schema) and plain ``DROP TABLE``; only
|
||||
# ``PRAGMA writable_schema=ON`` + sqlite_master surgery still work. Canonical
|
||||
# sessions/messages are intact; recovery rebuilds only the FTS layer.
|
||||
_MALFORMED_SCHEMA_MARKERS = ("malformed database schema",)
|
||||
_MALFORMED_DB_MARKERS = (*_MALFORMED_SCHEMA_MARKERS, "database disk image is malformed")
|
||||
|
||||
|
||||
def is_malformed_db_error(exc: BaseException) -> bool:
|
||||
"""Malformed-schema OR generic corrupt-image error. Diagnostics / offline
|
||||
recovery only — runtime repair must use :func:`is_malformed_schema_error`."""
|
||||
return isinstance(exc, sqlite3.DatabaseError) and any(
|
||||
marker in str(exc).lower() for marker in _MALFORMED_DB_MARKERS
|
||||
)
|
||||
|
||||
|
||||
# SQLITE_IOERR as a substring (wrapped strings still classify); shared by the
|
||||
# read-only open retry and the write-path BEGIN retry.
|
||||
_DISK_IO_ERROR_MARKER = "disk i/o error"
|
||||
|
||||
# "Store BUSY, not gone" — HTTP callers map these to 503 instead of 500.
|
||||
# Corruption deliberately absent: a malformed store must surface, not be
|
||||
# retried into a timeout.
|
||||
_TRANSIENT_SQLITE_MARKERS = (
|
||||
_DISK_IO_ERROR_MARKER, "database is locked", "database table is locked", "busy",
|
||||
)
|
||||
|
||||
|
||||
def _is_no_more_rows(exc: sqlite3.Error) -> bool:
|
||||
"""Transient engine error on contended WAL appends; the identical write succeeds
|
||||
standalone, so it retries like locked/busy. Message-scoped because some builds
|
||||
raise it as InterfaceError (outside DatabaseError)."""
|
||||
return "no more rows available" in str(exc).lower()
|
||||
|
||||
|
||||
def is_transient_sqlite_error(exc: BaseException) -> bool:
|
||||
""""Busy right now", not "damaged". One predicate so the read-only open
|
||||
retry and the HTTP 503-vs-500 split cannot drift apart."""
|
||||
return isinstance(exc, sqlite3.OperationalError) and any(
|
||||
marker in str(exc).lower() for marker in _TRANSIENT_SQLITE_MARKERS
|
||||
)
|
||||
|
||||
|
||||
def is_malformed_schema_error(exc: BaseException) -> bool:
|
||||
"""Only SQLite's explicit malformed-schema text. A generic "disk image is
|
||||
malformed" (SQLITE_CORRUPT) may be any B-tree/freelist page and does not
|
||||
prove canonical rows intact, so runtime repair must fail closed on it."""
|
||||
return isinstance(exc, sqlite3.DatabaseError) and any(
|
||||
marker in str(exc).lower() for marker in _MALFORMED_SCHEMA_MARKERS
|
||||
)
|
||||
|
||||
|
||||
# "Filesystem cannot accept another write" substrings (OSError, sqlite3, and
|
||||
# wrapped RPC strings all match the same helper).
|
||||
_DISK_FULL_MARKERS = (
|
||||
"no space left on device",
|
||||
"not enough space",
|
||||
"database or disk is full", # SQLITE_FULL
|
||||
"disk full",
|
||||
"full disk",
|
||||
"enospc",
|
||||
)
|
||||
|
||||
|
||||
def is_disk_full_error(exc: BaseException | str | None) -> bool:
|
||||
"""Disk-full / ENOSPC: OSError(ENOSPC), SQLITE_FULL, or matching strings."""
|
||||
if exc is None:
|
||||
return False
|
||||
if isinstance(exc, OSError) and getattr(exc, "errno", None) == errno.ENOSPC:
|
||||
return True
|
||||
lowered = (exc if isinstance(exc, str) else str(exc)).lower()
|
||||
return any(marker in lowered for marker in _DISK_FULL_MARKERS)
|
||||
|
||||
|
||||
# Every classify_persistence_error bucket; consumers enumerate this tuple so a
|
||||
# new bucket can never silently desynchronize them.
|
||||
PERSISTENCE_ERROR_CAUSES = (
|
||||
"locked", "compression", "compression_closed", "turn_lease", "corrupt", "replaced", "disk",
|
||||
"unknown",
|
||||
)
|
||||
|
||||
|
||||
# "Database FILE structurally damaged" substrings. NOTE: "database disk image is
|
||||
# malformed" contains "disk", so this check MUST run before the disk bucket in
|
||||
# classify_persistence_error or B-tree corruption reads as "free some disk space".
|
||||
_DB_CORRUPTION_MARKERS = (
|
||||
"malformed", # "database disk image is malformed" (SQLITE_CORRUPT)
|
||||
"file is not a database", # SQLITE_NOTADB (also connection-level poisoning)
|
||||
"not a database",
|
||||
"database corruption",
|
||||
)
|
||||
|
||||
|
||||
def classify_persistence_error(exc_or_str) -> str:
|
||||
"""Coarse cause bucket (PERSISTENCE_ERROR_CAUSES) so the user's guidance
|
||||
matches: "locked" = busy, retry; "disk" = full/read-only/permissions;
|
||||
"compression" = a live lease refused the write; "compression_closed" = adopt
|
||||
the rotated session id; "turn_lease" = fencing, not storage; "corrupt" =
|
||||
file damage (repair path, not disk space); "replaced" = stop writing."""
|
||||
if exc_or_str is None:
|
||||
return "unknown"
|
||||
# Lease refusals contain neither "locked" nor "busy": match by type, then by
|
||||
# phrase for strings that survived RPC wrapping.
|
||||
if isinstance(exc_or_str, SessionTurnLeaseLostError):
|
||||
return "turn_lease"
|
||||
if isinstance(exc_or_str, CompressionSessionClosedError):
|
||||
return "compression_closed"
|
||||
if isinstance(exc_or_str, CompressionSessionBusyError):
|
||||
return "compression"
|
||||
if isinstance(exc_or_str, StateDbReplacedError): # incl. DeletedWalGenerationError
|
||||
return "replaced"
|
||||
if isinstance(exc_or_str, StateDbCorruptError):
|
||||
return "corrupt"
|
||||
text = str(exc_or_str).lower()
|
||||
if "turn lease" in text:
|
||||
return "turn_lease"
|
||||
if "closed by compression" in text:
|
||||
return "compression_closed"
|
||||
if "being compressed" in text or "compression lease" in text:
|
||||
return "compression"
|
||||
if "was replaced underneath" in text:
|
||||
return "replaced"
|
||||
if "deleted state.db-wal" in text or "deleted state.db-shm" in text:
|
||||
return "replaced"
|
||||
# Corruption BEFORE the lock/disk buckets: "disk image is malformed"
|
||||
# contains "disk" and some wrapped strings mention "locked" recovery.
|
||||
if any(marker in text for marker in _DB_CORRUPTION_MARKERS):
|
||||
return "corrupt"
|
||||
if "locked" in text or "busy" in text:
|
||||
return "locked"
|
||||
if is_disk_full_error(exc_or_str) or "disk" in text or "readonly" in text or "read-only" in text:
|
||||
return "disk"
|
||||
return "unknown"
|
||||
|
||||
|
||||
class CompressionSessionClosedError(RuntimeError):
|
||||
"""A durable write targeted a parent already closed by compression."""
|
||||
|
||||
def __init__(self, session_id: str):
|
||||
self.session_id = session_id
|
||||
super().__init__(
|
||||
f"Session {session_id!r} is closed by compression; "
|
||||
"adopt its live continuation before appending messages"
|
||||
)
|
||||
|
||||
|
||||
class CompressionSessionBusyError(RuntimeError):
|
||||
"""A non-owner tried to write while compression owns the session."""
|
||||
|
||||
|
||||
class SessionCompressionInProgressError(CompressionSessionBusyError):
|
||||
"""A concurrent writer collided with a *live* compression lock — transient
|
||||
(the compressor publishes in seconds; ``_execute_write`` waits), unlike the
|
||||
parent class's other case (a compressor whose own lease is gone: permanent,
|
||||
fail fast). Subclassing keeps every existing handler working."""
|
||||
|
||||
|
||||
class SessionTurnLeaseLostError(RuntimeError):
|
||||
"""A transcript write presented a turn-lease holder that no longer owns it.
|
||||
Fail-fast fencing (no ``_execute_write`` retry): a later writer may already
|
||||
be persisting a newer turn, and landing this one would interleave a stale reply."""
|
||||
|
||||
|
||||
class StateDbReplacedError(RuntimeError):
|
||||
"""The state.db path no longer names the file this SessionDB opened
|
||||
(out-of-band cp/mv/restore). In-place FTS repair and fail-open trigger
|
||||
dropping cannot fix a generation mismatch; they amplify it."""
|
||||
|
||||
|
||||
class DeletedWalGenerationError(StateDbReplacedError):
|
||||
"""A live process holds a deleted state.db-wal / -shm generation. Opening or
|
||||
writing through this handle would mint a second WAL inode (split-brain ->
|
||||
intermittent SQLITE_CORRUPT / IOERR). Stop the writers; never unlink the WAL
|
||||
yourself. Subclasses StateDbReplacedError so every consumer that diverts
|
||||
transcripts on a replaced store handles this identically."""
|
||||
|
||||
|
||||
# SQLite header application_id (offset 68). Distinct from inode: ``cp`` onto the
|
||||
# same path keeps st_ino and truncates+rewrites.
|
||||
_STATE_DB_APPLICATION_ID_OFFSET = 68
|
||||
_STATE_DB_GENERATION_KEY = "db_file_generation"
|
||||
_STATE_DB_REPLACED_MSG = (
|
||||
"FATAL: state.db was replaced underneath the gateway; refusing further "
|
||||
"writes to this file. Divert transcripts to sessions/<id>.jsonl (and the "
|
||||
"gateway pending_messages spool) and restore or reopen after operator intervention."
|
||||
)
|
||||
_DELETED_WAL_GENERATION_MSG = (
|
||||
"FATAL: a live process holds a deleted state.db-wal or state.db-shm "
|
||||
"inode while the path names a different (or missing) generation. "
|
||||
"Refusing to open or write so a second WAL cannot be minted. "
|
||||
"Stop the gateway, dashboard, and cron writers that hold the deleted "
|
||||
"sidecar, then reopen. Do not delete the WAL yourself. "
|
||||
"database.journal_mode: delete is operator containment, not a new default."
|
||||
)
|
||||
|
||||
|
||||
class StateDbCorruptError(sqlite3.DatabaseError):
|
||||
"""A live SessionDB observed structural (non-FTS, non-replaced) corruption and
|
||||
is quarantined: sticky for the handle's life — writes fail fast, no reopen,
|
||||
no close-time checkpoint (a handle that kept writing after the first error
|
||||
checkpointed 15 pages under wrong page numbers and turned a readable file
|
||||
into "file is not a database"; SQLITE_DBCONFIG_NO_CKPT_ON_CLOSE on 3.12+
|
||||
also stops SQLite's own). Subclasses sqlite3.DatabaseError so every degrade
|
||||
path keeps working. Recovery boundary: restart on a repaired/restored file."""
|
||||
|
||||
|
||||
_STATE_DB_CORRUPT_MSG = (
|
||||
"FATAL: state.db reported structural corruption (database disk image is "
|
||||
"malformed outside the FTS shadow tables) on a live handle; refusing further "
|
||||
"writes, automatic reopen, and the close-time WAL checkpoint on this file. "
|
||||
"Stop the gateway, then run `hermes sessions recover --source <state.db> "
|
||||
"--inspect-only` or restore a snapshot. Unwritten transcripts are diverted to "
|
||||
"sessions/<id>.jsonl (and the gateway pending_messages spool)."
|
||||
)
|
||||
148
hermes_state_guard.py
Normal file
148
hermes_state_guard.py
Normal file
@@ -0,0 +1,148 @@
|
||||
"""Live-DB test-isolation guard and the per-process "last init error" record.
|
||||
Every SessionDB construction resolves its path through _ensure_test_isolation
|
||||
so a pytest-context process (env OR ancestry) can never open a production
|
||||
state.db; env-based so subprocess children are protected too."""
|
||||
|
||||
import os
|
||||
import sys
|
||||
import threading
|
||||
from pathlib import Path
|
||||
from typing import Any, Optional
|
||||
|
||||
try: # Hard dependency, but tolerate scaffold-phase imports before pip install.
|
||||
import psutil
|
||||
except ImportError: # pragma: no cover - stripped/scaffold installs only
|
||||
psutil = None # type: ignore[assignment]
|
||||
|
||||
# Field evidence: pytest fixture rows landed in the production state.db and a
|
||||
# pytest-spawned child flipped the journal mode under the live WAL writer,
|
||||
# destroying committed transcripts; any HERMES_HOME escape (fixture ordering, a
|
||||
# child spawned without it, a shell exporting the real home) fell through silently.
|
||||
|
||||
#: Env twin of ``_STATE_DB_GUARD_BYPASS`` for child processes (a module global
|
||||
#: cannot cross a process boundary, and ancestry arms the guard there).
|
||||
_STATE_DB_GUARD_BYPASS_ENV = "HERMES_STATE_DB_GUARD_BYPASS"
|
||||
|
||||
|
||||
def _real_platform_state_root() -> Optional[Path]:
|
||||
"""The REAL platform-default Hermes root. Avoids ``Path.home()`` /
|
||||
``hermes_constants``: tests monkeypatch Path.home to a tempdir while this
|
||||
module is imported lazily, which would misidentify the hermetic home as
|
||||
production or miss the real one. ``expanduser`` reads HOME/passwd, which the
|
||||
conftest never rewrites."""
|
||||
try:
|
||||
if sys.platform == "win32":
|
||||
base = os.environ.get("LOCALAPPDATA", "").strip()
|
||||
root = (
|
||||
Path(base) / "hermes"
|
||||
if base
|
||||
else Path(os.path.expanduser("~")) / "AppData" / "Local" / "hermes"
|
||||
)
|
||||
else:
|
||||
root = Path(os.path.expanduser("~")) / ".hermes"
|
||||
return root.resolve()
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
#: Exported by the hermetic conftest alongside the HERMES_HOME redirect (value:
|
||||
#: the isolation root). Unlike PYTEST_* (scrubbed by tests that rebuild a child
|
||||
#: env) it is OURS and inherits by default, so a child carrying it that resolves
|
||||
#: a production DB is by definition an isolation escape.
|
||||
_TEST_ISOLATION_MARKER_ENV = "HERMES_TEST_ISOLATION"
|
||||
|
||||
|
||||
def _running_under_pytest() -> bool:
|
||||
"""True when this process (or a parent test process) is a pytest run."""
|
||||
return bool(
|
||||
os.environ.get("PYTEST_CURRENT_TEST")
|
||||
or os.environ.get("PYTEST_VERSION")
|
||||
or os.environ.get(_TEST_ISOLATION_MARKER_ENV)
|
||||
)
|
||||
|
||||
|
||||
#: pytest launcher names, matched against each argv token's *basename* so
|
||||
#: ``/tmp/pytest-of-dev/...`` paths cannot false-positive.
|
||||
_PYTEST_LAUNCHER_NAMES = frozenset({"pytest", "py.test", "pytest.exe", "py.test.exe"})
|
||||
|
||||
#: Memoised ancestry answer: the tree above us doesn't change; keep the hot path free.
|
||||
_PYTEST_ANCESTOR: Optional[bool] = None
|
||||
|
||||
|
||||
def _process_looks_like_pytest(proc: Any) -> bool:
|
||||
"""True when *proc*'s command line is a pytest invocation (``pytest ...`` or
|
||||
``python -m pytest``). Unreadable cmdline => not pytest: guessing the other
|
||||
way would refuse production opens for unrelated reasons."""
|
||||
try:
|
||||
cmdline = proc.cmdline() or []
|
||||
except Exception:
|
||||
return False
|
||||
for arg in cmdline:
|
||||
try:
|
||||
# Split on both separators on every host: os.path.basename is
|
||||
# POSIX-only under Linux and would leave a Windows-style path
|
||||
# intact, making the matcher's answer depend on the platform.
|
||||
name = str(arg).strip('"').strip("'").replace("\\", "/").rsplit("/", 1)[-1].lower()
|
||||
except Exception:
|
||||
continue
|
||||
if name in _PYTEST_LAUNCHER_NAMES:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _has_pytest_ancestor() -> bool:
|
||||
"""True when an ancestor process is a pytest run. A child spawned with a
|
||||
rebuilt env loses PYTEST_* and the HERMES_HOME redirect together — aiming at
|
||||
production AND disarming the guard in one step; ancestry survives that.
|
||||
Fails open without psutil / on walk errors (never block real user runs)."""
|
||||
global _PYTEST_ANCESTOR
|
||||
if _PYTEST_ANCESTOR is not None:
|
||||
return _PYTEST_ANCESTOR
|
||||
found = False
|
||||
if psutil is not None:
|
||||
try:
|
||||
found = any(_process_looks_like_pytest(p) for p in psutil.Process().parents())
|
||||
except Exception:
|
||||
found = False
|
||||
_PYTEST_ANCESTOR = found
|
||||
return found
|
||||
|
||||
|
||||
def _in_test_context() -> bool:
|
||||
"""Test run by environment or ancestry. Env first (two dict lookups); the
|
||||
memoised ancestry walk runs at most once per real ``hermes`` invocation."""
|
||||
return _running_under_pytest() or _has_pytest_ancestor()
|
||||
|
||||
|
||||
def _is_production_state_db(resolved: Path, root: Path) -> bool:
|
||||
"""*resolved* is ``<root>/state.db`` or ``<root>/profiles/<name>/state.db``.
|
||||
Deeper scratch paths (repo worktrees under ~/.hermes/hermes-agent/...) are
|
||||
deliberately NOT matched so hermetic tests cannot false-positive."""
|
||||
if resolved.parent == root:
|
||||
return True
|
||||
try:
|
||||
parts = resolved.relative_to(root).parts
|
||||
except ValueError:
|
||||
return False
|
||||
return len(parts) == 3 and parts[0] == "profiles"
|
||||
|
||||
|
||||
# Last SessionDB() init error, per-process; surfaced by /resume-style slash
|
||||
# commands so users know WHY. Only SessionDB.__init__ writes it (kanban_db
|
||||
# failures are reported via their own callers, by design).
|
||||
_last_init_error: Optional[str] = None
|
||||
_last_init_error_lock = threading.Lock()
|
||||
|
||||
|
||||
def _set_last_init_error(msg: Optional[str]) -> None:
|
||||
"""Record (or clear with None) the most recent state.db init failure.
|
||||
__init__ only SETs on failure and never clears on success: a concurrent
|
||||
successful open would erase the cause another thread's /resume is about to format."""
|
||||
global _last_init_error
|
||||
with _last_init_error_lock:
|
||||
_last_init_error = msg
|
||||
|
||||
|
||||
def get_last_init_error() -> Optional[str]:
|
||||
"""Most recent state.db init failure (None if none/never attempted)."""
|
||||
return _last_init_error
|
||||
Reference in New Issue
Block a user